{"columns":["resultId","modelSlug","canonicalModelName","sourceVisibleModelName","configurationId","modelConfiguration","benchmarkSlug","benchmarkName","benchmarkCategory","benchmarkOrganisation","benchmarkVersion","score","normalizedScore","unit","scoreDirection","methodologyVersion","inclusionStatus","evidenceState","observedAt","publishedAt","sourceKey","sourceId","sourceTitle","sourcePublisher","sourceUrl","sourceDate","sourceCheckedAt","checkedAt","verificationStatus","notes"],"generatedAt":"2026-09-01","ledgerSchemaVersion":"1.0.0","projection":"public","recordCount":15598,"rows":[["evidence-2026-08-15-claude-fable-5-aa-automation-bench-1-0-6-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",46.2,46.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-claude-opus-4-8-aa-automation-bench-1-0-6-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",41,41,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-deepseek-v4-pro-0813-aa-automation-bench-1-0-6-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",43.2,43.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-glm-5-2-aa-automation-bench-1-0-6-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",26.2,26.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-glm-5-3-aa-automation-bench-1-0-6","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",48.2,48.2,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-gpt-5-6-sol-aa-automation-bench-1-0-6-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",45.8,45.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-kimi-k3-aa-automation-bench-1-0-6-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",46.7,46.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["evidence-2026-08-15-qwen-3-8-max-aa-automation-bench-1-0-6-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","1.0.6",39.8,39.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: AutomationBench v1.0.6 including PR #13."],["benchlm-ref-claude-fable-aaautomationbench-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.6,84.8708,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaautomationbench-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.6,79.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaautomationbench-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.6,79.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaautomationbench-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.5,84.5018,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaautomationbench-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.5,79,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaautomationbench-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",48.5,79,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaautomationbench-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",39.2,50.1845,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaautomationbench-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",39.2,32.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaautomationbench-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",39.2,32.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",37.5,43.9114,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",37.5,24,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",37.5,24,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.6,62.7306,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.6,49.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.6,49.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaautomationbench-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",32.7,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaautomationbench-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",32.7,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaautomationbench-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.1,92,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaautomationbench-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.1,92,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaautomationbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",27.8,8.1181,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaautomationbench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.1,60.8856,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaautomationbench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.1,47,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaautomationbench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.1,47,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.2,61.2546,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.2,47.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.2,47.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.2,94.4649,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.2,92.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.2,92.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",45.6,73.8007,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",45.6,64.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",45.6,64.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaautomationbench-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.4,95.203,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaautomationbench-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.4,93.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaautomationbench-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",51.4,93.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaautomationbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",52.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaautomationbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",52.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaautomationbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",52.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaautomationbench-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.8,63.4686,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaautomationbench-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.8,50.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaautomationbench-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",42.8,50.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaautomationbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis","2026",25.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-793","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,48.6,48.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-794","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,48.5,48.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-800","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,39.2,39.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-801","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,37.5,37.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-797","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.6,42.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-802","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,27.8,27.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-799","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.1,42.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-798","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.2,42.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-792","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,51.2,51.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-795","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,45.6,45.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-791","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,51.4,51.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1425","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,19.5527,19.5527,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1440","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,22.5286,22.5286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1951","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis AutomationBench-AA evaluation.",null,"Kimi K3; Artificial Analysis AutomationBench-AA evaluation.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,52.7,52.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Prefer independent AA AutomationBench 52.7 over provider-published 30.8 on the public subset."],["evidence-2026-07-1594","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,18.8277,18.8277,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-796","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.8,42.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1922","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.8,42.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1922--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,42.8,42.8,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1609","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,5.6706,5.6706,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1490","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,12.1956,12.1956,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-803","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aa-automation-bench","AA AutomationBench","agents","Artificial Analysis",null,25.6,25.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-2047","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis AA-Briefcase Elo.",null,"Artificial Analysis AA-Briefcase Elo.","aa-briefcase","AA-Briefcase","agents","Artificial Analysis","2026",63.4,63.4,"elo-proxy","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-Briefcase Elo 634; stored as Elo/10 display proxy, reference-only."],["evidence-2026-07-2007","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis AA-Briefcase Elo (mid).",null,"Artificial Analysis AA-Briefcase Elo (mid).","aa-briefcase","AA-Briefcase","agents","Artificial Analysis","2026",96.136,96.136,"elo-proxy","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-Briefcase Elo mid ≈ 961.36; stored as Elo/10 display proxy, reference-only."],["evidence-2026-07-1962","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis AA-Briefcase Elo.",null,"Kimi K3; Artificial Analysis AA-Briefcase Elo.","aa-briefcase","AA-Briefcase","agents","Artificial Analysis","2026",152.7,7.635,"elo-proxy","higher","1.4.1","reference-only","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","AA-Briefcase Elo ~1527 stored as Elo/10 display proxy only until fixed anchors are validated."],["evidence-2026-07-1935","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; AA-Briefcase Elo 863 reported via Artificial Analysis; stored as normalized display proxy (Elo/10) pending fixed transform.",null,"Muse Spark 1.1; AA-Briefcase Elo 863 reported via Artificial Analysis; stored as normalized display proxy (Elo/10) pending fixed transform.","aa-briefcase","AA-Briefcase","agents","Artificial Analysis","2026",86.3,86.3,"elo-proxy","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-briefcase","aa-briefcase","AA-Briefcase evaluation leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/aa-briefcase","2026-07-16","2026-07-16","2026-07-16","source-checked","Raw AA-Briefcase Elo is 863. Normalized as Elo/10 for registry visibility only; inclusion is reference-only until a fixed Elo transform is validated."],["evidence-2026-07-1935--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; AA-Briefcase Elo 863 reported via Artificial Analysis; stored as normalized display proxy (Elo/10) pending fixed transform.","muse-spark-1-1-xhigh","Muse Spark 1.1; AA-Briefcase Elo 863 reported via Artificial Analysis; stored as normalized display proxy (Elo/10) pending fixed transform.","aa-briefcase","AA-Briefcase","agents","Artificial Analysis","2026",86.3,86.3,"elo-proxy","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-briefcase","aa-briefcase","AA-Briefcase evaluation leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/aa-briefcase","2026-07-16","2026-07-16","2026-07-16","source-checked","Raw AA-Briefcase Elo is 863. Normalized as Elo/10 for registry visibility only; inclusion is reference-only until a fixed Elo transform is validated."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:androidworld:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",62,62,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-androidworld-2026-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",62,62,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-holo3-1-35b-a3b-androidworld-2026-07-21","holo3-1-35b-a3b","Holo3.1-35B-A3B","Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",79.3,84.1121,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-35b-a3b-androidworld-2026-07-27","holo3-1-35b-a3b","Holo3.1-35B-A3B","Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",79.3,84.1121,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-35b-a3b-androidworld-2026-08-01","holo3-1-35b-a3b","Holo3.1-35B-A3B","Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",79.3,84.1121,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-4b-androidworld-2026-07-21","holo3-1-4b","Holo3.1-4B","Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-4b-androidworld-2026-07-27","holo3-1-4b","Holo3.1-4B","Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-4b-androidworld-2026-08-01","holo3-1-4b","Holo3.1-4B","Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-9b-androidworld-2026-07-21","holo3-1-9b","Holo3.1-9B","Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-9b-androidworld-2026-07-27","holo3-1-9b","Holo3.1-9B","Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-1-9b-androidworld-2026-08-01","holo3-1-9b","Holo3.1-9B","Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",71,6.5421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-androidworld-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",70.3,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-androidworld-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",70.3,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-androidworld-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",70.3,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-androidworld-2026-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",70.3,70.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-qwen3-7-plus-androidworld-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-androidworld-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-androidworld-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:androidworld:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81,81,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-androidworld-2026-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81,81,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:androidworld:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81.9,81.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-androidworld-2026","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",81.9,81.9,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-androidworld:cell:vision:androidworld:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-androidworld","AndroidWorld","agents","Z.AI","2026",84.5,84.5,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-kimi-3-apexagents-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-apexagents","APEX-Agents","agents","Moonshot AI / APEX-Agents benchmark authors","2026",37.6,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-apexagents-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-apexagents","APEX-Agents","agents","Moonshot AI / APEX-Agents benchmark authors","2026",37.6,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-apexagents-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-apexagents","APEX-Agents","agents","Moonshot AI / APEX-Agents benchmark authors","2026",37.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-apexagentsaa-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33,69.6121,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-apexagentsaa-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33,69.6121,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-apexagentsaa-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33,69.6121,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.3,50.8621,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",12.2,24.7845,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",12.2,24.7845,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",12.2,24.7845,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",32,67.4569,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",32,67.4569,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",32,67.4569,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",47.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",47.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",47.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-apexagentsaa-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.5,29.7414,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-apexagentsaa-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.5,29.7414,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-apexagentsaa-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.5,29.7414,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-apexagentsaa-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.7,71.1207,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-apexagentsaa-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.7,71.1207,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-apexagentsaa-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.7,71.1207,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-apexagentsaa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.3,70.2586,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-apexagentsaa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.3,70.2586,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-apexagentsaa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",33.3,70.2586,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.2,59.2672,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.2,59.2672,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.2,59.2672,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.9,52.1552,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.9,52.1552,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",24.9,52.1552,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-apexagentsaa-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",37.7,79.7414,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-apexagentsaa-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",37.7,79.7414,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-apexagentsaa-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",37.7,79.7414,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",35.8,75.6466,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",35.8,75.6466,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",35.8,75.6466,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",38.9,82.3276,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",38.9,82.3276,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",38.9,82.3276,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-apexagentsaa-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",3.1,5.1724,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-apexagentsaa-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",3.1,5.1724,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-apexagentsaa-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",3.1,5.1724,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-apexagentsaa-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",0.7,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-apexagentsaa-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",0.7,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-apexagentsaa-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",0.7,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-apexagentsaa-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",17,35.1293,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-apexagentsaa-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",17,35.1293,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-apexagentsaa-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",17,35.1293,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-apexagentsaa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-apexagentsaa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-apexagentsaa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-apexagentsaa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-apexagentsaa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-apexagentsaa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",11.5,23.2759,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-apexagentsaa-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.5,59.9138,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-apexagentsaa-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.5,59.9138,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-apexagentsaa-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",28.5,59.9138,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-apexagentsaa-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",41.3,87.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-apexagentsaa-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",41.3,87.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-apexagentsaa-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",41.3,87.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",2.4,3.6638,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",2.4,3.6638,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",2.4,3.6638,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-apexagentsaa-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",10.6,21.3362,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-apexagentsaa-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",10.6,21.3362,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-apexagentsaa-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",10.6,21.3362,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-apexagentsaa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-apexagentsaa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-apexagentsaa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-apexagentsaa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-apexagentsaa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-apexagentsaa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",15.3,31.4655,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apexagentsaa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",22.4,46.7672,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apexagentsaa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",22.4,46.7672,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apexagentsaa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",22.4,46.7672,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-apexagentsaa-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.8,30.3879,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-apexagentsaa-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.8,30.3879,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-apexagentsaa-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","2026",14.8,30.3879,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-command-a-plus-apex-agents-standard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",1.6,1.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-apex-agents-standard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",13.2,13.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-apex-agents-standard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",13.4,13.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-apex-agents-standard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",6.1,6.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-apex-agents-standard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",2.4,2.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-apex-agents-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor","standard",16.6,16.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["xai-grok-4-6-release-apex-agents-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,59.2,59.2,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-1037","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.0383,33.0383,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1037--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.0383,33.0383,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1015","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,28.0236,28.0236,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1015--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,28.0236,28.0236,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1523","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,14.528,14.528,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1104","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,24.2625,24.2625,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1104--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,24.2625,24.2625,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1135","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,27.7286,27.7286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1076","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,12.1681,12.1681,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1219","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,32.0059,32.0059,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-920","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,47.0501,47.0501,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-920--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,47.0501,47.0501,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1031","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,14.4543,14.4543,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1256","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.7021,33.7021,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1256--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.7021,33.7021,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1052","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.2596,33.2596,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1052--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,33.2596,33.2596,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1288","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,28.1711,28.1711,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1288--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,28.1711,28.1711,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1150","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,24.9263,24.9263,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1150--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,24.9263,24.9263,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-975","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,37.6844,37.6844,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-975--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,37.6844,37.6844,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["xai-grok-4-6-release-apex-agents-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,56.7,56.7,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-1115","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,17.0354,17.0354,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["xai-grok-4-6-release-apex-agents-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,47.1,47.1,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-apex-agents-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,57.5,57.5,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["evidence-2026-07-1899","hy3","Hy3","tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.",null,"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,25.6,25.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-06","2026-07-06","production::tencent-hy3-hf","tencent-hy3-hf","Tencent Hy3 model card on Hugging Face","Tencent Hy Team","https://huggingface.co/tencent/Hy3","2026-07-06","2026-07-16","2026-07-16","provider-reported","APEX-Agents score listed in Hugging Face evaluation results for tencent/Hy3."],["evidence-2026-07-1186","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,11.5044,11.5044,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1416","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,28.4661,28.4661,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1940","kimi-k3","Kimi K3","APEX-Agents; max reasoning; provider-published Kimi K3 evaluation.",null,"APEX-Agents; max reasoning; provider-published Kimi K3 evaluation.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,37.6,37.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 37.6 on APEX-Agents."],["evidence-2026-07-1940--configuration--kimi-k3-max","kimi-k3","Kimi K3","APEX-Agents; max reasoning; provider-published Kimi K3 evaluation.","kimi-k3-max","APEX-Agents; max reasoning; provider-published Kimi K3 evaluation.","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,37.6,37.6,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 37.6 on APEX-Agents."],["evidence-2026-07-931","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,2.4336,2.4336,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1060","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,10.6195,10.6195,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["evidence-2026-07-1825","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,1.8437,1.8437,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1481","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,15.3392,15.3392,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1204","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","apex-agents","APEX-Agents-AA","agents","Artificial Analysis / Mercor",null,22.4189,22.4189,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-apex-agents","aa-apex-agents","APEX-Agents-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/apex-agents-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","APEX-Agents-AA success from Artificial Analysis."],["benchlm-ref-claude-fable-aaagenticindex-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",52.81,97.7852,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaagenticindex-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",52.81,95.5446,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaagenticindex-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",52.81,95.5446,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaagenticindex-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.39,82.1143,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaagenticindex-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.39,80.2328,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaagenticindex-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.39,80.2328,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaagenticindex-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.18,87.3069,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaagenticindex-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.18,85.3064,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaagenticindex-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.18,85.3064,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaagenticindex-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",55.26,100,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaagenticindex-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",55.26,100,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaagenticindex-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",46.69,86.3949,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaagenticindex-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",46.69,84.4153,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaagenticindex-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",46.69,84.4153,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaagenticindex-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",9.16,16.5457,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaagenticindex-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",9.16,16.1666,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaagenticindex-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",9.16,16.1666,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaagenticindex-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.58,2.4381,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaagenticindex-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.58,2.3823,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaagenticindex-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.58,2.3823,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,51.9263,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,51.9263,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,50.7365,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,50.7365,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,50.7365,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaagenticindex-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.17,50.7365,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,57.305,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,55.992,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,55.992,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,57.305,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,55.992,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",31.06,55.992,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,63.5771,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,63.5771,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,62.1204,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,62.1204,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,62.1204,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaagenticindex-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",34.43,62.1204,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,67.1692,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,65.6301,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,65.6301,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,67.1692,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,65.6301,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",36.36,65.6301,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",7.1,12.7117,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",7.1,12.4204,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",7.1,12.4204,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",6.17,10.9808,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",6.17,10.7292,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",6.17,10.7292,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.4,39.3263,"index","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.4,38.4252,"index","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.4,38.4252,"index","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.45,69.1978,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.45,67.6123,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.45,67.6123,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",26.82,49.4137,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",26.82,48.2815,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",26.82,48.2815,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",38.72,71.5615,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",38.72,69.9218,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",38.72,69.9218,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaagenticindex-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.27,0,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaagenticindex-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.27,0,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaagenticindex-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.27,0,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaagenticindex-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",7.93,13.9298,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaagenticindex-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",7.93,13.9298,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",10.97,19.9144,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",10.97,19.4581,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",10.97,19.4581,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaagenticindex-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",14.45,26.3912,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaagenticindex-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",14.45,25.7865,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaagenticindex-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",14.45,25.7865,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaagenticindex-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.51,2.255,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaagenticindex-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.51,2.255,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaagenticindex-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.79,2.7641,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaagenticindex-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.79,2.7641,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaagenticindex-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.39,46.7523,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaagenticindex-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.39,45.681,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaagenticindex-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.39,45.681,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaagenticindex-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.87,55.0903,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaagenticindex-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.87,53.828,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaagenticindex-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.87,53.828,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaagenticindex-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",43.06,79.6389,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaagenticindex-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",43.06,77.8141,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaagenticindex-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",43.06,77.8141,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.72,2.6987,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.72,2.6368,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.72,2.6368,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.17,1.675,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.17,1.6367,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.17,1.6367,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaagenticindex-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.96,1.2842,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaagenticindex-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.96,1.2548,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaagenticindex-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",0.96,1.2548,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,47.3479,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,46.263,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,46.263,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,47.3479,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,46.263,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaagenticindex-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.71,46.263,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaagenticindex-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.01,38.6004,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaagenticindex-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.01,37.7159,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaagenticindex-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.01,37.7159,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaagenticindex-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",41.08,75.9538,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaagenticindex-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",41.08,74.2135,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaagenticindex-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",41.08,74.2135,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.17,55.6486,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.17,54.3735,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.17,54.3735,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.54,50.7538,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.54,49.5908,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.54,49.5908,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaagenticindex-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.87,83.0076,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaagenticindex-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.87,81.1057,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaagenticindex-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",44.87,81.1057,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.6,84.3663,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.6,82.4332,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.6,82.4332,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",54,100,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",54,97.7087,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",54,97.7087,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.38,87.6791,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.38,85.6701,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",47.38,85.6701,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaagenticindex-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",13.17,24.0089,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaagenticindex-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",13.17,23.4588,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaagenticindex-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",13.17,23.4588,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaagenticindex-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.1,5.2671,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaagenticindex-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.1,5.1464,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaagenticindex-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.1,5.1464,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaagenticindex-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",24.1,44.3514,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaagenticindex-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",24.1,43.3352,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaagenticindex-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",24.1,43.3352,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaagenticindex-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.69,84.5338,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaagenticindex-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.69,82.5968,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaagenticindex-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",45.69,82.5968,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaagenticindex-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,56.6909,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaagenticindex-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,55.3919,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaagenticindex-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,55.3919,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaagenticindex-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,56.6909,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaagenticindex-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,55.3919,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaagenticindex-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.73,55.3919,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaagenticindex-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",32.34,59.6873,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaagenticindex-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",32.34,58.3197,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaagenticindex-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",32.34,58.3197,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaagenticindex-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",8,14.0571,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaagenticindex-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",8,14.0571,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaagenticindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,39.866,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaagenticindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,38.9525,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaagenticindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,38.9525,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaagenticindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,39.866,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaagenticindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,38.9525,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaagenticindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.69,38.9525,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaagenticindex-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.27,55.8347,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaagenticindex-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.27,54.5554,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaagenticindex-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.27,54.5554,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.59,54.5691,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.59,53.3188,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.59,53.3188,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaagenticindex-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",50.07,92.6857,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaagenticindex-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",50.07,90.5619,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaagenticindex-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",50.07,90.5619,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaagenticindex-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",2.25,3.6851,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaagenticindex-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",2.25,3.6007,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaagenticindex-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",2.25,3.6007,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaagenticindex-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.31,1.9356,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaagenticindex-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.31,1.8913,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaagenticindex-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.31,1.8913,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaagenticindex-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.1,1.5448,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaagenticindex-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.1,1.5094,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaagenticindex-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.1,1.5094,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaagenticindex-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",11.99,21.8128,"index","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaagenticindex-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",11.99,21.313,"index","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaagenticindex-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",11.99,21.313,"index","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.11,53.6758,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.11,52.4459,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",29.11,52.4459,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaagenticindex-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.58,47.1059,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaagenticindex-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.58,46.0266,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaagenticindex-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",25.58,46.0266,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaagenticindex-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",35.36,65.308,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaagenticindex-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",35.36,63.8116,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaagenticindex-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",35.36,63.8116,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaagenticindex-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",5.52,9.7711,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaagenticindex-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",5.52,9.5472,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaagenticindex-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",5.52,9.5472,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19,34.8595,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19,34.0607,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19,34.0607,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaagenticindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.2449,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaagenticindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.056,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaagenticindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.056,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaagenticindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.2449,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaagenticindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.056,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaagenticindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",4.7,8.056,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaagenticindex-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.69,52.8941,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaagenticindex-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.69,51.6821,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaagenticindex-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",28.69,51.6821,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaagenticindex-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.54,69.3653,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaagenticindex-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.54,67.776,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaagenticindex-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",37.54,67.776,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.99,3.2012,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.99,3.1278,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",1.99,3.1278,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.36,50.4188,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.36,49.2635,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.36,49.2635,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaagenticindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,36.4415,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaagenticindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,35.6065,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaagenticindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,35.6065,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaagenticindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,36.4415,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaagenticindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,35.6065,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaagenticindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",19.85,35.6065,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.72,38.0607,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.72,37.1886,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.72,37.1886,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaagenticindex-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.03,49.8046,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaagenticindex-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.03,48.6634,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaagenticindex-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.03,48.6634,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaagenticindex-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.55,50.7724,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaagenticindex-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.55,49.609,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaagenticindex-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",27.55,49.609,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.41,39.3449,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.41,38.4434,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.41,38.4434,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaagenticindex-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.59,56.4303,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaagenticindex-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.59,55.1373,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaagenticindex-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",30.59,55.1373,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaagenticindex-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.81,38.2282,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaagenticindex-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.81,37.3522,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaagenticindex-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",20.81,37.3522,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaagenticindex-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.53,39.5682,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaagenticindex-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.53,38.6616,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaagenticindex-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",21.53,38.6616,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaagenticindex-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.65,6.1466,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaagenticindex-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.65,6.1466,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaagenticindex-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.65,6.1466,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaagenticindex-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aaagenticindex","Artificial Analysis Agentic Index","agents","Artificial Analysis","2026",3.65,6.1466,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aabriefcaseelo-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1574,100,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aabriefcaseelo-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1574,86.5314,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aabriefcaseelo-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1574,87.8738,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["xai-grok-4-6-release-aa-briefcase-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1574,98.863636,"elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1347,78.7453,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1346,65.4982,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1346,68.9369,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aabriefcaseelo-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1720,100,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aabriefcaseelo-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1720,100,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1388,82.5843,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1386,69.1882,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1386,72.2591,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",831,30.4307,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",833,18.1734,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",833,26.3289,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",831,30.4307,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",833,18.1734,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",833,26.3289,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",932,39.8876,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",931,27.214,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",930,34.3854,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",932,39.8876,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",931,27.214,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",930,34.3854,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",634,11.985,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",636,0,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",636,9.9668,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",961,42.603,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",964,30.2583,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",964,37.2093,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aabriefcaseelo-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1260,70.5993,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aabriefcaseelo-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1254,57.0111,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aabriefcaseelo-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1254,61.2957,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aabriefcaseelo-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1154,60.6742,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aabriefcaseelo-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1153,47.6937,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1501,93.1648,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1505,80.1661,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1503,81.9767,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["xai-grok-4-6-release-aa-briefcase-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1502,71.590909,"elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["benchlm-ref-grok-4-5-aabriefcaseelo-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1323,76.4981,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aabriefcaseelo-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1318,62.9151,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aabriefcaseelo-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1317,66.5282,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["xai-grok-4-6-release-aa-briefcase-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1313,0,"elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-aa-briefcase-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1577,100,"elo","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["benchlm-ref-inkling-aabriefcaseelo-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",836,30.8989,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aabriefcaseelo-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",839,18.7269,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aabriefcaseelo-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",839,26.8272,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aabriefcaseelo-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1543,97.0974,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aabriefcaseelo-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1541,83.4871,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aabriefcaseelo-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1540,85.0498,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",873,34.3633,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",878,22.3247,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",878,30.0664,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aabriefcaseelo-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1110,56.5543,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aabriefcaseelo-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1109,43.6347,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aabriefcaseelo-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",1110,49.3355,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aabriefcaseelo-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",506,0,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aabriefcaseelo-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",516,0,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",863,33.427,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",868,21.4022,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",868,29.2359,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",870,34.0824,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",873,21.8635,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",873,29.6512,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",908,37.6404,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",912,25.4613,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aabriefcaseelo","Artificial Analysis Briefcase","agents","Artificial Analysis","2026",912,32.8904,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaenterpriseopsgym-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",51.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaenterpriseopsgym-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",51.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaenterpriseopsgym-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",51.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44,72.2656,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44,68.8596,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44,72.2656,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44.7,75,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44.7,71.9298,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",44.7,75,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,55.0781,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,49.5614,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,55.0781,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,55.0781,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,49.5614,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",39.6,55.0781,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,58.2031,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,53.0702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,58.2031,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,58.2031,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,53.0702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.4,58.2031,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaenterpriseopsgym-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.2,65.2344,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaenterpriseopsgym-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.2,60.9649,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",50.1,96.0938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",50.1,95.614,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",50.1,96.0938,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.3,10.9375,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.3,10.9375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.7,67.1875,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.7,63.1579,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.7,67.1875,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",46.6,82.4219,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",46.6,80.2632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",46.6,82.4219,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaenterpriseopsgym-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.9,64.0351,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaenterpriseopsgym-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",42.9,67.9687,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaenterpriseopsgym-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",25.5,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaenterpriseopsgym-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",25.5,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.8,59.7656,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.8,54.8246,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",40.8,59.7656,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaenterpriseopsgym-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",38.1,42.9825,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaenterpriseopsgym-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",38.1,49.2188,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaenterpriseopsgym-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45.3,77.3437,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaenterpriseopsgym-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45.3,74.5614,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaenterpriseopsgym-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45.3,77.3437,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",32.1,25.7813,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",32.1,16.6667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",32.1,25.7813,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",33.7,32.0313,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",33.7,23.6842,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",33.7,32.0313,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaenterpriseopsgym-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",47.2,82.8947,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaenterpriseopsgym-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",47.2,84.7656,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.9,13.2812,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.9,2.6316,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",28.9,13.2812,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45,76.1719,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45,73.2456,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaenterpriseopsgym","Artificial Analysis EnterpriseOps-Gym","agents","Artificial Analysis","2026",45,76.1719,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaharveylab-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.6,97.1989,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaharveylab-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.6,97.1989,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaharveylab-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.6,98.7608,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaharveylab-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91.1,90.1961,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaharveylab-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91.1,90.1961,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaharveylab-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91.1,95.6629,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaharveylab-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",90.1,87.395,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaharveylab-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",90.1,87.395,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaharveylab-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",90.1,94.4238,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,62.7451,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,62.7451,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,83.5192,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,62.7451,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,62.7451,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.3,83.5192,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,71.4286,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,71.4286,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,87.3606,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,71.4286,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,71.4286,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",84.4,87.3606,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaharveylab-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",58.9,0,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaharveylab-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",58.9,0,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaharveylab-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",47.2,41.2639,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaharveylab-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91,89.916,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaharveylab-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91,89.916,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaharveylab-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",91,95.539,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaharveylab-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",86.3,76.7507,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaharveylab-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",86.3,76.7507,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaharveylab-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.9,81.2325,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaharveylab-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.9,81.2325,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaharveylab-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.9,91.6976,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaharveylab-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.2,79.2717,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaharveylab-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.2,79.2717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaharveylab-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",87.2,90.8302,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaharveylab-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",85.2,73.6695,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaharveylab-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",85.2,73.6695,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaharveylab-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",85.2,88.3519,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaharveylab-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",13.9,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaharveylab-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",92.4,93.8375,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaharveylab-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",92.4,93.8375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaharveylab-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",92.4,97.2739,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaharveylab-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",94.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaharveylab-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",94.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",73.3,40.3361,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",73.3,40.3361,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",73.3,73.6059,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaharveylab-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",88.4,82.6331,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaharveylab-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",88.4,82.6331,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaharveylab-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",88.4,92.3172,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",69.1,28.5714,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",69.1,28.5714,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",69.1,68.4015,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaharveylab-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.1,95.7983,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaharveylab-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.1,95.7983,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaharveylab-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",93.1,98.1413,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaharveylab-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.7,63.8655,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaharveylab-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.7,63.8655,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaharveylab-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",81.7,84.0149,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaharveylab-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",83.4,68.6275,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaharveylab-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",83.4,68.6275,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaharveylab-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","2026",83.4,86.1214,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaharveylab-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-aaharveylab","Artificial Analysis Harvey LAB-AA","agents","Artificial Analysis","Harvey Lab-AA criterion pass rate as of 2026-07-23",94.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports Harvey Lab-AA criterion pass rate=94.6, cites Artificial Analysis as of 2026-07-23, and applies the all-results max-effort rule. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-claude-opus-4-7-adaptive-aaitbench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",46.7,81.2253,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaitbench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",46.7,81.2253,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaitbench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",46.7,81.2253,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",31.5,51.1858,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.3,64.6245,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaitbench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",30.3,48.8142,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaitbench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",30.3,48.8142,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaitbench-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",37.3,62.6482,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaitbench-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",37.3,62.6482,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaitbench-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",37.3,62.6482,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaitbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.7,73.3202,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaitbench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.7,73.3202,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaitbench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.7,73.3202,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaitbench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",45.8,79.4466,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaitbench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",45.8,79.4466,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaitbench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",45.8,79.4466,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaitbench-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",40.3,68.5771,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaitbench-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",40.3,68.5771,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaitbench-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",40.3,68.5771,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaitbench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",56.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaitbench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",56.2,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaitbench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",56.2,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaitbench-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",51,89.7233,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaitbench-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",51,89.7233,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaitbench-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",51,89.7233,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaitbench-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",5.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaitbench-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",5.6,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaitbench-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",5.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaitbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",47.7,83.2016,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaitbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",47.7,83.2016,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaitbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",47.7,83.2016,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaitbench-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.2,64.4269,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaitbench-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.2,64.4269,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaitbench-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",38.2,64.4269,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaitbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.5,72.9249,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaitbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.5,72.9249,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaitbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaitbench","Artificial Analysis ITBench-AA","agents","Artificial Analysis","2026",42.5,72.9249,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aatau3banking-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,65.2632,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aatau3banking-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,60.9467,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aatau3banking-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,65.2632,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aatau3banking-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.6,69.4737,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aatau3banking-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.6,65.6805,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aatau3banking-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.6,69.4737,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aatau3banking-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",30.3,81.6568,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aatau3banking-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",30.3,83.6842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aatau3banking-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",28.2,72.6316,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aatau3banking-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",28.2,69.2308,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aatau3banking-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",28.2,72.6316,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,44.7368,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,37.8698,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,44.7368,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,44.7368,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,37.8698,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",22.9,44.7368,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,60,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,55.0296,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,60,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,60,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,55.0296,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.8,60,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aatau3banking-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",16.5,11.0526,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aatau3banking-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",16.5,0,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",16.5,11.0526,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",16.5,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",16.5,11.0526,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aatau3banking-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",24.5,53.1579,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aatau3banking-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",24.5,47.3373,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aatau3banking-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",24.5,53.1579,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aatau3banking-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",15.1,3.6842,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aatau3banking-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",15.1,3.6842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aatau3banking-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,65.2632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aatau3banking-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,60.9467,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aatau3banking-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",26.8,65.2632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aatau3banking-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",31.3,88.9474,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aatau3banking-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",31.3,87.574,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aatau3banking-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.2,67.3684,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aatau3banking-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.2,63.3136,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aatau3banking-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",27.2,67.3684,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aatau3banking-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33,97.8947,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aatau3banking-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33,97.6331,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aatau3banking-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33,97.8947,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aatau3banking-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",31.8,91.5789,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aatau3banking-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",31.8,90.5325,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aatau3banking-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",31.8,91.5789,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aatau3banking-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",32.6,95.7895,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aatau3banking-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",32.6,95.2663,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aatau3banking-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",32.6,95.7895,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aatau3banking-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",23.7,48.9474,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aatau3banking-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",23.7,42.6036,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aatau3banking-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",23.7,48.9474,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aatau3banking-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aatau3banking-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33.4,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aatau3banking-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",33.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aatau3banking-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",14.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aatau3banking-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",14.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aatau3banking-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.2,56.8421,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aatau3banking-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.2,51.4793,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aatau3banking-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aatau3banking","Artificial Analysis Tau3-Banking","agents","Artificial Analysis","2026",25.2,56.8421,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2065","claude-opus-5","Claude Opus 5","AutomationBench (Zapier-style business workflows) as published in the Anthropic Claude Opus 5 launch table. Not the AA AutomationBench scale.",null,"AutomationBench (Zapier-style business workflows) as published in the Anthropic Claude Opus 5 launch table. Not the AA AutomationBench scale.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",26,26,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 26.0% pass rate. Kept on benchlm-automationbench only — AA AutomationBench uses a different score scale."],["benchlm-ref-claude-opus-5-automationbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",26,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-automationbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",26,15.7895,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-automationbench-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",25.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-automationbench-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",25.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-automationbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",30.8,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-automationbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",30.8,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-automationbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026",30.8,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-flash-vision-exp-automationbench-public-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","benchlm-automationbench","AutomationBench","agents","Moonshot AI","2026 public set",25.7,25.7,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value explicitly labelled as the 600-task public set. Benchmark-owner documentation confirms strict task pass rate as the metric; the frozen registry retains this provider run as reference-only."],["deepseek-v4-pro-0813-release-automationbench-public-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",29.1,29.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",27.2,27.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",10.8,10.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",25.1,25.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",12.8,12.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",31.8,31.8,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",12.9,12.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-automationbench-public-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-automationbench","AutomationBench","agents","Moonshot AI","Public 600-task set, August 2026",30.8,30.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["google-gemini-37-eval-automationbench-private-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","automationbench-private","AutomationBench Private Set","agents","AutomationBench","Private set, August 2026",10.7,10.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-automationbench-private-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","automationbench-private","AutomationBench Private Set","agents","AutomationBench","Private set, August 2026",17,17,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-automationbench-private-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","automationbench-private","AutomationBench Private Set","agents","AutomationBench","Private set, August 2026",30.4,30.4,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-automationbench-private-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","automationbench-private","AutomationBench Private Set","agents","AutomationBench","Private set, August 2026",23.6,23.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-minimax-m3-bankertoolbench-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-bankertoolbench","BankerToolBench","agents","MiniMax","2026",76.1,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-bankertoolbench-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-bankertoolbench","BankerToolBench","agents","MiniMax","2026",76.1,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-bankertoolbench-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-bankertoolbench","BankerToolBench","agents","MiniMax","2026",76.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-393","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",76.84,76.84,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-388","claude-mythos-5","Claude Mythos 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",77.09,77.09,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-520","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",55.56,55.56,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-453","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",58.87,58.87,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-437","claude-opus-4-7","Claude Opus 4.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",66.06,66.06,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-403","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",68.06,68.06,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-508","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",53.31,53.31,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-494","claude-sonnet-5","Claude Sonnet 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",66.49,66.49,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-553","deepseek-v4-pro","DeepSeek V4 Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",52.21,52.21,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-547","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",25.41,25.41,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-466","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",59.9,59.9,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-420","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",49.1,49.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-542","gemma-4-31b","Gemma 4 31B","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",27.31,27.31,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-488","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",54.85,54.85,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-459","glm-5-1","GLM-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",49.36,49.36,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-530","glm-5-2","GLM-5.2","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",60.98,60.98,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-514","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",56.85,56.85,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-483","gpt-5-3-codex","GPT-5.3-Codex","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",61.78,61.78,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-426","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",59.82,59.82,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-477","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",54.2,54.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-414","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",69.7,69.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-535","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","gpt-5-5-pro-default-high","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",60.37,60.37,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-472","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",64.62,64.62,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-398","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",75.49,75.49,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-432","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",67.83,67.83,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-498","grok-4-3","Grok 4.3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",35.09,35.09,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-412","grok-4-5","Grok 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",62.09,62.09,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-446","mimo-v2-5-pro","MiMo-V2.5-Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",43.12,43.12,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-526","minimax-m2-7","MiniMax M2.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",35.38,35.38,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-449","minimax-m3","MiniMax M3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",49.75,49.75,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-440","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",59.81,59.81,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-408","muse-spark-1-1","Muse Spark 1.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",70.87,70.87,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-502","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-agents","BenchLM Agentic prior","agents","BenchLM","bench-align-v5.1",55.68,55.68,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (agentic) used to estimate missing category coverage under methodology 1.3.0."],["benchlm-ref-lfm2-5-230m-bfclv4-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.03,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-bfclv4-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.03,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-bfclv4-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.03,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",49.73,53.1777,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",49.73,53.1777,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",49.73,53.1777,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.08,0.0926,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.08,0.0926,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",21.08,0.0926,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-07-21","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",44.2,42.9313,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-07-27","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",44.2,42.9313,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",44.2,42.9313,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-07-21","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",45.6,45.5253,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-07-27","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",45.6,45.5253,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",45.6,45.5253,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bfclv4-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",25.15,7.6339,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bfclv4-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",25.15,7.6339,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bfclv4-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",25.15,7.6339,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-bfclv4-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",75,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-bfclv4-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",75,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-bfclv4-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",75,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-bfclv4-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",72.9,96.1089,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-bfclv4-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",72.9,96.1089,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-bfclv4-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",72.9,96.1089,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-bfclv4-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",39.22,33.7039,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-bfclv4-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",39.22,33.7039,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-bfclv4-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-bfclv4","Berkeley Function Calling Leaderboard v4","agents","Arcee AI","2026",39.22,33.7039,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-357","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","berkeley-function-calling-leaderboard","Berkeley Function-Calling Leaderboard","agents","Berkeley Gorilla","V3",75,75,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-358","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","berkeley-function-calling-leaderboard","Berkeley Function-Calling Leaderboard","agents","Berkeley Gorilla","V3",72.9,72.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["benchlm-ref-claude-opus-4-5-claweval-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.6,75.5587,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-claweval-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.6,75.5587,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-claweval-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.6,75.5587,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-claweval-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",70.4,90.6425,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-claweval-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",70.4,90.6425,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-claweval-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",70.4,90.6425,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-claweval-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.8,87.0112,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-claweval-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.8,87.0112,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-claweval-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.8,87.0112,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-claweval-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",40.2,48.4637,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-claweval-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",40.2,48.4637,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-claweval-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",40.2,48.4637,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-claweval-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-claweval-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-claweval-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-claweval-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.8,75.838,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-claweval-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.8,75.838,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-claweval-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",59.8,75.838,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-claweval-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",49.2,61.0335,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-claweval-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",49.2,61.0335,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-claweval-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",49.2,61.0335,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-claweval-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-claweval-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-claweval-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-claweval-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.7,72.905,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-claweval-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.7,72.905,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-claweval-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.7,72.905,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-claweval-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",55.8,70.2514,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-claweval-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",55.8,70.2514,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-claweval-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",55.8,70.2514,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-claweval-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-claweval-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-claweval-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-claweval-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",53.8,67.4581,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-claweval-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",53.8,67.4581,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-claweval-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",53.8,67.4581,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-claweval-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",60.3,76.5363,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-claweval-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",60.3,76.5363,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-claweval-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",60.3,76.5363,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-claweval-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",52.3,65.3631,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-claweval-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",52.3,65.3631,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-claweval-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",52.3,65.3631,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-claweval-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-claweval-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-claweval-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-claweval-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",45.2,55.4469,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-claweval-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",45.2,55.4469,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-claweval-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",45.2,55.4469,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-claweval-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-claweval-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-claweval-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",57.8,73.0447,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-claweval-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-claweval-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-claweval-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.3,79.3296,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-claweval-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-claweval-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-claweval-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-claweval-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",48.7,60.3352,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-claweval-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",48.7,60.3352,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-claweval-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",48.7,60.3352,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-claweval-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",74.5,96.3687,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-claweval-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",74.5,96.3687,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-claweval-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",74.5,96.3687,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-claweval-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-claweval-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-claweval-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.8,81.4246,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-super-100b-claweval-2026-07-21","nemotron-3-super-100b","Nemotron 3 Super 100B","Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",5.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-super-100b-claweval-2026-07-27","nemotron-3-super-100b","Nemotron 3 Super 100B","Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",5.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-super-100b-claweval-2026-08-01","nemotron-3-super-100b","Nemotron 3 Super 100B","Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",5.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-claweval-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",69.8,89.8045,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-claweval-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",69.8,89.8045,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-claweval-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",69.8,89.8045,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-claweval-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",77.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-claweval-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",77.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-claweval-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",77.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-claweval-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.1,80.4469,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-claweval-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.1,80.4469,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-claweval-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",63.1,80.4469,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-claweval-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",56.8,71.648,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-claweval-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",56.8,71.648,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-claweval-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",56.8,71.648,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-claweval-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",72.4,93.4358,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-claweval-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",72.4,93.4358,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-claweval-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",72.4,93.4358,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-claweval-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",58.8,74.4413,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-claweval-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",58.8,74.4413,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-claweval-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",58.8,74.4413,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-claweval-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",68.7,88.2682,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-claweval-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",68.7,88.2682,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-claweval-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",68.7,88.2682,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-claweval-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",65.2,83.3799,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-claweval-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",65.2,83.3799,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-claweval-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",65.2,83.3799,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-claweval-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.7,79.8883,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-claweval-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.7,79.8883,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-claweval-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",62.7,79.8883,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-claweval-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.1,86.0335,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-claweval-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.1,86.0335,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-claweval-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","2026",67.1,86.0335,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-average:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",54.7,54.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-claweval-average-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",54.7,54.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-claweval-average-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",50.4,50.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-average:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",60.1,60.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-claweval-average-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",60.1,60.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-average:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",56.9,56.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-claweval-average","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Average",56.9,56.9,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-pass3:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",52.5,52.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-claweval-pass-3-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",52.5,52.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-claweval-pass-3-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",42.6,42.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-pass3:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",57.4,57.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-claweval-pass-3-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",57.4,57.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:claweval-pass3:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",57.4,57.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","ClawEval-MM reports both Pass@3 and Average tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-claweval-pass-3","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-claweval","Claw-Eval","agents","Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","Pass@3",57.4,57.4,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:cowork:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","cowork-bench","CoWorkBench","agents","Qwen","2026-08",68.2,68.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","CoWorkBench is Qwen's in-house long-horizon office-work benchmark spanning multiple domains. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-cowork-bench-2026-08-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","cowork-bench","CoWorkBench","agents","Qwen","2026-08",68.2,68.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:cowork:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","cowork-bench","CoWorkBench","agents","Qwen","2026-08",45.1,45.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","CoWorkBench is Qwen's in-house long-horizon office-work benchmark spanning multiple domains. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen3-6-27b-cowork-bench-2026-08-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","cowork-bench","CoWorkBench","agents","Qwen","2026-08",61,61,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:cowork:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","cowork-bench","CoWorkBench","agents","Qwen","2026-08",65.1,65.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","CoWorkBench is Qwen's in-house long-horizon office-work benchmark spanning multiple domains. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-cowork-bench-2026-08-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","cowork-bench","CoWorkBench","agents","Qwen","2026-08",65.1,65.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:cowork:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","cowork-bench","CoWorkBench","agents","Qwen","2026-08",70.7,70.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","CoWorkBench is Qwen's in-house long-horizon office-work benchmark spanning multiple domains. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-cowork-bench-2026-08","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","cowork-bench","CoWorkBench","agents","Qwen","2026-08",70.7,70.7,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-sakana-fugu-cyber-ctirealm-2026-07-21","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","benchlm-ctirealm","CTI-REALM","agents","Sakana AI","2026",72.1,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-cyber-ctirealm-2026-07-27","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","benchlm-ctirealm","CTI-REALM","agents","Sakana AI","2026",72.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-cyber-ctirealm-2026-08-01","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","benchlm-ctirealm","CTI-REALM","agents","Sakana AI","2026",72.1,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybench-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybench","Cybench","agents","Stanford / Cybench authors","2025",92.9,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybench-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybench","Cybench","agents","Stanford / Cybench authors","2025",92.9,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybench-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybench","Cybench","agents","Stanford / Cybench authors","2025",92.9,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1274","claude-mythos-5","Claude Mythos 5","Claude Mythos Preview; unguided; 35-problem subset (system card via Cybench leaderboard note ⁶)",null,"Claude Mythos Preview; unguided; 35-problem subset (system card via Cybench leaderboard note ⁶)","cybench","Cybench","agents","Stanford / Cybench authors",null,100,100,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved; subset size noted by maintainers."],["evidence-2026-07-1277","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5; unguided; 39-problem subset (system card via Cybench leaderboard note ³)",null,"Claude Opus 4.5; unguided; 39-problem subset (system card via Cybench leaderboard note ³)","cybench","Cybench","agents","Stanford / Cybench authors",null,82,82,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved; subset size noted by maintainers."],["evidence-2026-07-1276","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6; unguided; 37-problem subset (system card via Cybench leaderboard note ⁵)",null,"Claude Opus 4.6; unguided; 37-problem subset (system card via Cybench leaderboard note ⁵)","cybench","Cybench","agents","Stanford / Cybench authors",null,93,93,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved; subset size noted by maintainers."],["evidence-2026-07-1275","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7; unguided; 35-problem subset (system card via Cybench leaderboard note ⁸)",null,"Claude Opus 4.7; unguided; 35-problem subset (system card via Cybench leaderboard note ⁸)","cybench","Cybench","agents","Stanford / Cybench authors",null,96,96,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved; subset size noted by maintainers."],["evidence-2026-07-1279","grok-4-1","Grok 4.1","Grok 4.1 Thinking; unguided (xAI model card via Cybench leaderboard note ⁴)",null,"Grok 4.1 Thinking; unguided (xAI model card via Cybench leaderboard note ⁴)","cybench","Cybench","agents","Stanford / Cybench authors",null,39,39,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved for Grok 4.1 Thinking."],["evidence-2026-07-1278","muse-spark","Muse Spark","Muse Spark; unguided (Meta safety report via Cybench leaderboard note ⁷)",null,"Muse Spark; unguided (Meta safety report via Cybench leaderboard note ⁷)","cybench","Cybench","agents","Stanford / Cybench authors",null,65.4,65.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::cybench-official","cybench-official","Cybench official leaderboard","Cybench authors","https://cybench.github.io/","2026-07-15","2026-07-15","2026-07-15","official-leaderboard","Official Cybench leaderboard unguided % solved."],["evidence-2026-07-1908","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Cybench).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Cybench).","cybench","Cybench","agents","Stanford / Cybench authors",null,92.9,92.9,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Cybench; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1908--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Cybench).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Cybench).","cybench","Cybench","agents","Stanford / Cybench authors",null,92.9,92.9,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Cybench; retained with provider-reported provenance via Meta evaluation report."],["benchlm-ref-claude-opus-4-5-cybergym-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",50.6,16.9336,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-cybergym-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",50.6,16.9336,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-cybergym-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",50.6,16.9336,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-cybergym-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",66.6,53.5469,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-cybergym-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",66.6,53.5469,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-cybergym-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",66.6,53.5469,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-cybergym-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",73.1,68.4211,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-cybergym-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",73.1,68.4211,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-cybergym-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",73.1,68.4211,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-cybergym-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",65.2,50.3432,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-cybergym-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",65.2,50.3432,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-cybergym-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",65.2,50.3432,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-cybergym-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",76.7,76.659,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-cybergym-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",76.7,76.659,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-cyber-cybergym-2026-07-21","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",86.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-cyber-cybergym-2026-07-27","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",86.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-cyber-cybergym-2026-08-01","sakana-fugu-cyber","Fugu Cyber","Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",86.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-cyber-cybergym-2026-07-27","gemini-3-5-flash-cyber","Gemini 3.5 Flash Cyber","Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",83.2,91.5332,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-cyber-cybergym-2026-08-01","gemini-3-5-flash-cyber","Gemini 3.5 Flash Cyber","Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",83.2,91.5332,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-cybergym-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.2,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-cybergym-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.2,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-cybergym-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.2,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-cybergym-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",68.7,58.3524,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-cybergym-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",68.7,58.3524,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-cybergym-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",68.7,58.3524,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-cybergym-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",79,81.9222,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-cybergym-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",79,81.9222,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-cybergym-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",79,81.9222,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-cybergym-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-cybergym-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-cybergym-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-cybergym-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",77.9,79.405,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-cybergym-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",77.9,79.405,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-cybergym-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",77.9,79.405,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-cybergym-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",84.5,94.508,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-cybergym-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",84.5,94.508,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-cybergym-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",84.5,94.508,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-cybergym-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-cybergym-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-cybergym-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",81.8,88.3295,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-cybergym-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.5,0.6865,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-cybergym-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.5,0.6865,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-cybergym-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",43.5,0.6865,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybergym-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",59,36.1556,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybergym-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",59,36.1556,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-cybergym-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","cybergym","CyberGym","agents","UC Berkeley SunBlaze","2026",59,36.1556,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-fable-5-cybergym-rolling-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",83.8,83.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["deepseek-v4-pro-0813-release-cybergym-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",83.1,83.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-cybergym-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",78.3,78.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-08-15-claude-opus-4-8-cybergym-rolling-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",78.1,78.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["deepseek-v4-pro-0813-release-cybergym-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",38.7,38.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-cybergym-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",76.7,76.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-cybergym-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",52.7,52.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-cybergym-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",83.3,83.3,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["evidence-2026-08-15-deepseek-v4-pro-0813-cybergym-rolling-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",83.3,83.3,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["evidence-2026-08-15-glm-5-2-cybergym-rolling-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",77.2,77.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["evidence-2026-08-15-glm-5-3-cybergym-rolling","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",84.5,84.5,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["evidence-2026-08-15-gpt-5-6-sol-cybergym-rolling-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",83.6,83.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["deepseek-v4-pro-0813-release-cybergym-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",80,80,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-08-15-kimi-k3-cybergym-rolling-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",80,80,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["evidence-2026-08-15-qwen-3-8-max-cybergym-rolling-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","cybergym","CyberGym","agents","UC Berkeley SunBlaze","rolling",78.5,78.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: no web tools, Pass@1 over 1,507 tasks, domain whitelist."],["deepseek-v4-flash-0731-cybergym-2026-07-31","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-0731-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","cybergym","CyberGym","agents","UC Berkeley SunBlaze","standard",76.7,76.7,"percent","higher","2.3.0","reference-only","direct","2026-07-31","2026-07-31","production::deepseek-v4-flash-0731-model-card","deepseek-v4-flash-0731-model-card","DeepSeek-V4-Flash-0731 official model card and provider evaluation","DeepSeek","https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","2026-07-31","2026-08-24","2026-08-24","provider-reported","Source-native DeepSeek subject value with the official code-agent configuration; the frozen CyberGym provider track remains reference-only."],["deepseek-v4-flash-vision-exp-cybergym-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","cybergym","CyberGym","agents","UC Berkeley SunBlaze","standard",75.3,75.3,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value. Canonical CyberGym identity, configuration, metric and unit are resolved; the frozen protocol registry keeps this provider track reference-only."],["benchlm-ref-claude-opus-5-draco-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-draco","Data Research and Analysis with Complex Operations","agents","Anthropic","2026",88.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-draco-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-draco","Data Research and Analysis with Complex Operations","agents","Anthropic","2026",88.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deckbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-deckbench","DECK-Bench (Internal)","agents","Moonshot AI","2026",73.5,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deckbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-deckbench","DECK-Bench (Internal)","agents","Moonshot AI","2026",73.5,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deckbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-deckbench","DECK-Bench (Internal)","agents","Moonshot AI","2026",73.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-deepplanning-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",26.4,25.0522,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-deepplanning-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",26.4,25.0522,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-deepplanning-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",26.4,25.0522,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-deepplanning-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.6,0.4175,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-deepplanning-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.6,0.4175,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-deepplanning-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.6,0.4175,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepplanning-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.4,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepplanning-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.4,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepplanning-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",14.4,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-deepplanning-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",37.6,48.4342,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-deepplanning-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",37.6,48.4342,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-deepplanning-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",37.6,48.4342,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-deepplanning-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",41.5,56.5762,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-deepplanning-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",41.5,56.5762,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-deepplanning-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",41.5,56.5762,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",25.9,24.0084,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",25.9,24.0084,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",25.9,24.0084,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-deepplanning-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",62.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-deepplanning-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",62.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-deepplanning-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-deepplanning","DeepPlanning","agents","DeepPlanning authors","2026",62.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-deepsearchqa-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.7,33.8509,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-deepsearchqa-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.7,33.8509,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-deepsearchqa-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.7,33.8509,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-deepsearchqa-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",93.1,94.0994,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-deepsearchqa-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",93.1,94.0994,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-deepsearchqa-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",93.1,94.0994,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-deepsearchqa-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",95,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-deepsearchqa-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",95,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",69.7,21.4286,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",69.7,21.4286,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",69.7,21.4286,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-deepsearchqa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.6,33.5404,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-deepsearchqa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.6,33.5404,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-deepsearchqa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",73.6,33.5404,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-deepsearchqa-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",62.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-deepsearchqa-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",62.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-deepsearchqa-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",62.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepsearchqa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",77.1,44.4099,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepsearchqa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",77.1,44.4099,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-deepsearchqa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",77.1,44.4099,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-deepsearchqa-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.5,92.236,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-deepsearchqa-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.5,92.236,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-deepsearchqa-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.5,92.236,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deepsearchqa-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",95,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deepsearchqa-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",95,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-deepsearchqa-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",74.8,37.2671,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-deepsearchqa-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",74.8,37.2671,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-deepsearchqa-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",74.8,37.2671,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepsearchqa-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",84.9,68.6335,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepsearchqa-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",84.9,68.6335,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepsearchqa-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",84.9,68.6335,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-deepsearchqa-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.82,93.2298,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-deepsearchqa-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.82,93.2298,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-deepsearchqa-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","2026",92.82,93.2298,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-benchlm-deepsearchqa-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","900 questions",74.6,74.6,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only because this is the existing source-specific DeepSearchQA registry lane."],["evidence-2026-08-muse-glimmer-30b-benchlm-deepsearchqa-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","900 questions",74.6,74.6,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only because this is the existing source-specific DeepSearchQA registry lane."],["benchlm-ref-kimi-3-deepsearchqa-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-deepsearchqa","DeepSearchQA","agents","Meta AI","DeepSearchQA F1",95,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports DeepSearchQA F1=95.0 under its all-results max-effort rule. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-inkling-designarenaagenticwebdev-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenaagenticwebdev","Design Arena Agentic Web Dev Elo","agents","Design Arena / Intelligence","2026",1257,50,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-designarenaagenticwebdev-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenaagenticwebdev","Design Arena Agentic Web Dev Elo","agents","Design Arena / Intelligence","2026",1257,50,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-designarenaagenticwebdev-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenaagenticwebdev","Design Arena Agentic Web Dev Elo","agents","Design Arena / Intelligence","2026",1257,50,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-912","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,51.1191,51.1191,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-912--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,51.1191,51.1191,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1196","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,29.3942,29.3942,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1178","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,43.957,43.957,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1178--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,43.957,43.957,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-962","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,44.6732,44.6732,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-962--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,44.6732,44.6732,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1163","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,39.5703,39.5703,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1163--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,39.5703,39.5703,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1107","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.4357,40.4357,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1107--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.4357,40.4357,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1078","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,28.0215,28.0215,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1222","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,42.1665,42.1665,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-923","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,50.0746,50.0746,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-923--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,50.0746,50.0746,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1232","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,28.3497,28.3497,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1259","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,42.7335,42.7335,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1259--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,42.7335,42.7335,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1291","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,34.6165,34.6165,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1291--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,34.6165,34.6165,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1153","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,31.4831,31.4831,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1153--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,31.4831,31.4831,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-978","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,46.6428,46.6428,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-978--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,46.6428,46.6428,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1118","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,26.2608,26.2608,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1142","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.8236,40.8236,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1419","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,38.5258,38.5258,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1435","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.2268,40.2268,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1952","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis EnterpriseOps-Gym-AA evaluation.",null,"Kimi K3; Artificial Analysis EnterpriseOps-Gym-AA evaluation.","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,45.3,45.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis EnterpriseOps-Gym-AA score for Kimi K3."],["evidence-2026-07-1587","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.4357,40.4357,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1007","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,32.1098,32.1098,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1247","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,33.7213,33.7213,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1604","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,28.8869,28.8869,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1484","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,30.8863,30.8863,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1128","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,45.0015,45.0015,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-07-1206","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","enterpriseops-gym-aa","EnterpriseOps-Gym-AA","agents","Artificial Analysis / ServiceNow",null,40.5551,40.5551,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-enterpriseops-gym","aa-enterpriseops-gym","EnterpriseOps-Gym-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","EnterpriseOps-Gym-AA success from Artificial Analysis."],["evidence-2026-08-15-claude-fable-5-exploitbench-2-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",78,78,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-07-886","claude-opus-4-7","Claude Opus 4.7","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","exploitbench","ExploitBench","agents","ExploitBench authors","2",27,27,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-exploitbench","benchlm-exploitbench","ExploitBench via BenchLM","BenchLM / ExploitBench","https://benchlm.ai/benchmarks/exploitBench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-claude-opus-4-8-exploitbench-2-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",40,40,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-08-15-glm-5-2-exploitbench-2-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",24.4,24.4,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-08-15-glm-5-3-exploitbench-2","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","exploitbench","ExploitBench","agents","ExploitBench authors","2",54.4,54.4,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-07-885","gpt-5-5","GPT-5.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","exploitbench","ExploitBench","agents","ExploitBench authors","2",34,34,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-exploitbench","benchlm-exploitbench","ExploitBench via BenchLM","BenchLM / ExploitBench","https://benchlm.ai/benchmarks/exploitBench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-gpt-5-6-sol-exploitbench-2-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",76.5,76.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-08-15-kimi-k3-exploitbench-2-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",32.2,32.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["evidence-2026-08-15-qwen-3-8-max-exploitbench-2-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","exploitbench","ExploitBench","agents","ExploitBench authors","2",28.8,28.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: 300 rounds, average coverage over 41 tasks and 3 revisions."],["benchlm-ref-claude-mythos-5-exploitgym-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",17.5,50.7599,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-exploitgym-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",17.5,50.7599,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-exploitgym-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",17.5,50.7599,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-exploitgym-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",6,15.8055,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-exploitgym-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",6,15.8055,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-exploitgym-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",6,15.8055,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-exploitgym-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",13.4,38.2979,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-exploitgym-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",13.4,38.2979,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-exploitgym-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",13.4,38.2979,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-exploitgym-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",12.4,35.2584,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-exploitgym-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",12.4,35.2584,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-exploitgym-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",12.4,35.2584,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitgym-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",33.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitgym-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",33.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitgym-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",33.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitgym-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",23.2,68.0851,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitgym-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",23.2,68.0851,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitgym-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",23.2,68.0851,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-exploitgym-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",0.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-exploitgym-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",0.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-exploitgym-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","exploitgym","ExploitGym","agents","ExploitGym authors","2026",0.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-353","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",17.5,17.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-356","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",6,6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-354","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",13.4,13.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-355","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",12.4,12.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-351","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",33.7,33.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-352","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",23.2,23.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-576","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",0.8,0.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1909","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (ExploitGym).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (ExploitGym).","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",0.8,0.8,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for ExploitGym; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1909--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (ExploitGym).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (ExploitGym).","exploitgym","ExploitGym","agents","ExploitGym authors","2026-05",0.8,0.8,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for ExploitGym; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-08-15-claude-fable-5-exploitgym-2h-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",181,20.83,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-claude-opus-4-8-exploitgym-2h-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",80,9.21,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-glm-5-2-exploitgym-2h-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",29,3.34,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-glm-5-3-exploitgym-2h","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","exploitgym","ExploitGym","agents","ExploitGym authors","2h",105,12.08,"tasks","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-gpt-5-6-sol-exploitgym-2h-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",216,24.86,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-kimi-k3-exploitgym-2h-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",36,4.14,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-qwen-3-8-max-exploitgym-2h-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","2h",14,1.61,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 2-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-claude-fable-5-exploitgym-6h-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",247,28.42,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-claude-opus-4-8-exploitgym-6h-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",120,13.81,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-glm-5-2-exploitgym-6h-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",39,4.49,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-glm-5-3-exploitgym-6h","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","exploitgym","ExploitGym","agents","ExploitGym authors","6h",130,14.96,"tasks","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-gpt-5-6-sol-exploitgym-6h-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",293,33.72,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-kimi-k3-exploitgym-6h-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",70,8.06,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-qwen-3-8-max-exploitgym-6h-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","exploitgym","ExploitGym","agents","ExploitGym authors","6h",26,2.99,"tasks","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI 6-hour budget, single-run Pass@1 on 869 tasks. Normalized as tasks/869."],["evidence-2026-08-15-longcat-2-0-forte-2026-08","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","forte","FORTE","agents","AGI-Eval-Official","2026-08",73.2,73.2,"percent","higher","2.1.0","reference-only","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 in-house FORTE score."],["evidence-2026-07-2054","claude-opus-5","Claude Opus 5","Frontier-Bench v0.1; mini-SWE-agent harness; GKE backend; mean reward over 5 attempts per task; Opus 4.8 safety-classifier fallback. Peak table score as published.",null,"Frontier-Bench v0.1; mini-SWE-agent harness; GKE backend; mean reward over 5 attempts per task; Opus 4.8 safety-classifier fallback. Peak table score as published.","frontier-bench","Frontier-Bench","agents","Frontier-Bench","v0.1",43.3,43.3,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 43.3% on Frontier-Bench v0.1 (SOTA in the launch table). Family is reference-only until multi-lab weight review."],["benchlm-ref-claude-opus-5-frontierbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierbench","FrontierBench v0.1","agents","Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute","2026",43.3,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-frontierbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierbench","FrontierBench v0.1","agents","Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute","2026",43.3,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-873","claude-fable-5","Claude Fable 5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,52.3,52.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-872","claude-mythos-5","Claude Mythos 5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,52.3,52.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-875","claude-opus-4-6","Claude Opus 4.6","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,47.8,47.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-877","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,45.5,45.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-882","gemini-3-flash","Gemini 3 Flash","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,35.2,35.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-879","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,38.5,38.5,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-876","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,46.1,46.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-883","glm-5","GLM-5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,31.5,31.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-874","gpt-5-4","GPT-5.4","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,48.2,48.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-878","grok-4-1","Grok 4.1","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,39.7,39.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-881","kimi-k2-5","Kimi K2.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,36.5,36.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-884","mistral-large-3","Mistral Large 3","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,30.8,30.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-880","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","gaia","GAIA","agents","GAIA authors",null,37.4,37.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-gaia","benchlm-gaia","GAIA leaderboard via BenchLM","BenchLM / GAIA","https://benchlm.ai/benchmarks/gaia","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["benchlm-ref-minimax-m3-gdpvalrubrics-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalrubrics","GDPval rubrics","agents","MiniMax","2026",74.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalrubrics-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalrubrics","GDPval rubrics","agents","MiniMax","2026",74.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalrubrics-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalrubrics","GDPval rubrics","agents","MiniMax","2026",74.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaanormalized-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",62.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaanormalized-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",62.3,91.6176,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaanormalized-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",62.3,91.4831,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaanormalized-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.8,79.8077,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaanormalized-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.7,73.0882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaanormalized-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.7,72.9809,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.7,87.6603,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.6,80.2941,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.6,80.1762,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdpvalaanormalized-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",68,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdpvalaanormalized-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",68.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",55.4,88.7821,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",55.2,81.1765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",55.2,81.0573,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaanormalized-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",10.7,17.1474,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaanormalized-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",10.9,16.0294,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaanormalized-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",10.9,16.0059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,51.9231,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,51.9231,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,47.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,47.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,47.5771,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaanormalized-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.4,47.5771,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,55.1282,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.514,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,55.1282,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.514,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",39.9,63.9423,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",39.9,63.9423,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40,58.8235,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40,58.8235,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40,58.7372,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaanormalized-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40,58.7372,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.4,64.7436,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.3,59.2647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.3,59.1777,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.4,64.7436,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.3,59.2647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",40.3,59.1777,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",8.3,13.3013,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",8.6,12.6471,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",8.6,12.6285,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7.1,11.3782,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7.4,10.8824,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7.3,10.7195,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.3,37.3397,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.2,34.1176,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.2,34.0675,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",42.4,67.9487,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",42.2,62.0588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",42.2,61.9677,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32,51.2821,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.9118,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.8429,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",46.1,73.8782,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",46.2,67.9412,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",46.2,67.8414,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gdpvalaanormalized-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7.6,11.1765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gdpvalaanormalized-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7.6,11.1601,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",13.1,20.9936,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",13.5,19.8529,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",13.5,19.8238,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15.2,24.359,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15.5,22.7941,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15.5,22.7606,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gdpvalaanormalized-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gdpvalaanormalized-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gdpvalaanormalized-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gdpvalaanormalized-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaanormalized-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.3,53.3654,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaanormalized-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.3,48.9706,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaanormalized-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.3,48.8987,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaanormalized-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",37.8,60.5769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaanormalized-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",37.8,55.5882,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaanormalized-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",37.8,55.5066,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaanormalized-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",50.7,81.25,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaanormalized-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",50.5,74.2647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaanormalized-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",50.5,74.1557,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0.1,0.1603,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0.4,0.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0.4,0.5874,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.7,45.9936,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.9,42.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.9,42.4376,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.7,45.9936,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.9,42.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",28.9,42.4376,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",24.4,39.1026,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",24.4,35.8824,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",24.4,35.8297,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.7,71.6346,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.6,65.5882,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.6,65.4919,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.6,53.8462,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.5,49.2647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.5,49.1924,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",30,48.0769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",30.1,44.2647,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",30.1,44.1997,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.5,79.3269,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.6,72.9412,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",49.6,72.8341,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.2,86.859,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.1,79.5588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.1,79.442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",61.8,99.0385,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",61.8,90.8824,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",61.8,90.7489,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.1,86.6987,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.1,79.5588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",54.1,79.442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15,24.0385,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15.1,22.2059,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",15.1,22.1733,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3,4.8077,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.4,5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.4,4.9927,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaanormalized-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",29.2,46.7949,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaanormalized-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",29.2,42.9412,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaanormalized-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",29.2,42.8781,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaanormalized-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",51.7,82.8526,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaanormalized-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",51.4,75.5882,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaanormalized-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",51.4,75.4772,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaanormalized-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.7,57.2115,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaanormalized-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.8,52.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaanormalized-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.8,52.5698,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaanormalized-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.7,57.2115,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaanormalized-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.8,52.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaanormalized-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",35.8,52.5698,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaanormalized-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",36.9,59.1346,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaanormalized-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",36.8,54.1176,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaanormalized-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",36.8,54.0382,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-gdpvalaanormalized-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.9,7.2059,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-gdpvalaanormalized-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.9,7.1953,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaanormalized-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.4,40.7051,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaanormalized-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.1,36.9118,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaanormalized-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.1,36.8576,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.4,40.7051,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.1,36.9118,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.1,36.8576,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.5,55.2885,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.4,50.514,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.3,54.9679,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.3,50.4412,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",34.3,50.3671,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaanormalized-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",59,94.5513,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaanormalized-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",59.3,87.2059,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaanormalized-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",59.4,87.2247,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",2.2,3.5256,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",2.5,3.6765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",2.5,3.6711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",16.7,26.7628,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",16.9,24.8529,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",16.9,24.8164,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.3,61.3782,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.3,56.3235,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.3,56.2408,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.9,52.7244,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.9,48.3824,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.9,48.3113,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaanormalized-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.7,71.6346,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaanormalized-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.5,65.4412,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaanormalized-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",44.5,65.3451,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",6.6,10.5769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7,10.2941,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",7,10.279,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",21.4,34.2949,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",21.6,31.7647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",21.6,31.7181,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaanormalized-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.4,7.0513,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaanormalized-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.6,6.7647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaanormalized-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.6,6.7548,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.4,7.0513,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.6,6.7647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",4.6,6.7548,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaanormalized-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.2,51.6026,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaanormalized-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.2,47.3529,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaanormalized-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32.2,47.2834,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",43.7,70.0321,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",43.8,64.4118,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",43.8,64.3172,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.2,53.2051,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.1,48.6765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",33.1,48.605,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaanormalized-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,37.0192,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaanormalized-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,33.9706,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaanormalized-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,33.9207,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,37.0192,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,33.9706,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.1,33.9207,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",23.9,38.3013,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",24.1,35.4412,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",24.1,35.3891,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",32,51.2821,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.9118,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.8429,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.8,50.9615,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.9118,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",31.9,46.8429,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",27.4,43.9103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",27.6,40.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",27.6,40.5286,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.7,62.0192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.6,56.7647,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",38.6,56.6814,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",21.8,34.9359,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",22.1,32.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",22.1,32.4523,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.9,41.5064,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.8,37.9412,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",25.8,37.8855,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",2.7,4.3269,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.2,4.7059,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.2,4.699,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",2.7,4.3269,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.2,4.7059,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gdpvalaanormalized","GDPval-AA normalized","agents","Artificial Analysis","2026",3.2,4.699,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaa-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1748,100,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaa-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1747,94.2424,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-gdpvalaa-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1747,94.1949,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaa-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1495,86.6279,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaa-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1493,81.4141,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gdpvalaa-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1493,81.373,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaa-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1594,91.8605,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaa-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1593,86.4646,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gdpvalaa-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1593,86.421,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdpvalaa-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1861,100,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdpvalaa-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1862,100,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaa-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1607,92.5476,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaa-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1603,86.9697,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-gdpvalaa-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1603,86.9258,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaa-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",714,45.3488,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaa-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",718,42.2727,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-gdpvalaa-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",718,42.2514,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaa-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",217,19.0803,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaa-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",231,17.6768,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gdpvalaa-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",231,17.6678,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,68.2347,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,68.2347,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,63.9394,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,63.9394,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,63.9071,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gdpvalaa-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1147,63.9071,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,70.4545,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,66.0606,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,66.0273,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,70.4545,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,66.0606,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,66.0273,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,76.2685,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,76.2685,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,71.6162,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,71.6162,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,71.58,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gdpvalaa-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1299,71.58,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1307,76.6913,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1306,71.9697,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1306,71.9334,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1307,76.6913,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1306,71.9697,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1306,71.9334,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",665,42.759,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",672,39.9495,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",672,39.9293,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",642,41.5433,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",649,38.7879,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",647,38.6673,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",965,58.6152,"elo","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",965,54.7475,"elo","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",965,54.7198,"elo","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1349,78.9112,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1344,73.8889,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1345,73.9021,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1140,67.8647,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1139,63.5354,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1139,63.5033,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1421,82.7167,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1423,77.8788,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1423,77.8395,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaa-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",-144,0,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaa-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",-119,0,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-gdpvalaa-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",-119,0,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gdpvalaa-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",651,38.8889,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gdpvalaa-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",651,38.8693,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",761,47.833,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",770,44.899,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",770,44.8763,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaa-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",804,50.1057,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaa-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",811,46.9697,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gdpvalaa-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",811,46.946,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gdpvalaa-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",88,10.4545,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gdpvalaa-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",88,10.4493,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gdpvalaa-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",231,17.6768,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gdpvalaa-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",231,17.6678,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaa-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1165,69.186,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaa-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1166,64.899,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gdpvalaa-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1166,64.8662,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaa-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1257,74.0486,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaa-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1256,69.4444,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gdpvalaa-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1256,69.4094,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaa-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1514,87.6321,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaa-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1510,82.2727,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gdpvalaa-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1510,82.2312,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",503,34.1966,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",508,31.6667,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",508,31.6507,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",41,9.778,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",63,9.1919,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",63,9.1873,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaa-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",226,19.556,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaa-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",241,18.1818,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-gdpvalaa-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",241,18.1726,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1075,64.4292,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1079,60.5051,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1079,60.4745,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1075,64.4292,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1079,60.5051,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-gdpvalaa-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1079,60.4745,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaa-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",987,59.778,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaa-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",988,55.9091,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gdpvalaa-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",988,55.8809,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1395,81.3425,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1392,76.3131,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gdpvalaa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1392,76.2746,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1171,69.5032,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1169,65.0505,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1169,65.0177,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1100,65.7505,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1101,61.6162,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1101,61.5851,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaa-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1490,86.3636,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaa-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1491,81.3131,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gdpvalaa-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1491,81.2721,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1584,91.3319,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1582,85.9091,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1582,85.8657,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1736,99.3658,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1736,93.6869,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1735,93.5891,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1581,91.1734,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1583,85.9596,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1583,85.9162,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaa-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",799,49.8414,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaa-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",802,46.5152,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gdpvalaa-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",802,46.4917,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaa-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",559,37.1564,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaa-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",567,34.6465,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-gdpvalaa-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",567,34.629,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaa-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1085,64.9577,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaa-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1084,60.7576,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gdpvalaa-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1084,60.7269,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaa-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1535,88.7421,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaa-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1527,83.1313,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-gdpvalaa-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1528,83.1398,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaa-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1214,71.7759,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaa-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1215,67.3737,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-gdpvalaa-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1215,67.3397,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaa-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1214,71.7759,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaa-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1215,67.3737,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gdpvalaa-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1215,67.3397,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaa-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1239,73.0973,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaa-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1237,68.4848,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gdpvalaa-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1237,68.4503,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-gdpvalaa-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",598,36.2121,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-gdpvalaa-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",598,36.1938,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1009,60.9408,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1003,56.6667,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gdpvalaa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1003,56.6381,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1009,60.9408,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1003,56.6667,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gdpvalaa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1003,56.6381,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaa-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1189,70.4545,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaa-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1188,66.0101,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gdpvalaa-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1188,65.9768,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1187,70.3488,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1186,65.9091,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1186,65.8758,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaa-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1679,96.3531,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaa-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1686,91.1616,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gdpvalaa-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1687,91.1661,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaa-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",545,36.4165,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaa-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",550,33.7879,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gdpvalaa-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",550,33.7708,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaa-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",-16,6.7653,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaa-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",7,6.3636,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-gdpvalaa-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",7,6.3604,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaa-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",90,12.3679,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaa-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",111,11.6162,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-gdpvalaa-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",111,11.6103,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaa-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",833,51.6385,"elo","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaa-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",838,48.3333,"elo","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gdpvalaa-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",838,48.3089,"elo","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1265,74.4715,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1265,69.899,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1265,69.8637,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaa-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1158,68.8161,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaa-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1159,64.5455,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gdpvalaa-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1159,64.5129,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaa-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1395,81.3425,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaa-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1391,76.2626,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-gdpvalaa-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1391,76.2241,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaa-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",633,41.0677,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaa-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",640,38.3333,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-gdpvalaa-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",640,38.314,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",929,56.7125,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",933,53.1313,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",933,53.1045,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaa-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",588,38.6892,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaa-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",592,35.9091,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-gdpvalaa-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",592,35.891,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaa-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",588,38.6892,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaa-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",592,35.9091,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-gdpvalaa-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",592,35.891,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaa-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1144,68.0761,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaa-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1143,63.7374,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gdpvalaa-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1143,63.7052,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaa-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1374,80.2326,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaa-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1375,75.4545,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-gdpvalaa-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1375,75.4165,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",484,33.1924,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",492,30.8586,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",492,30.843,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",467,32.2939,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",465,29.4949,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",465,29.4801,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1164,69.1332,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1162,64.697,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1162,64.6643,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,58.4567,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,54.596,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-gdpvalaa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,54.5684,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,58.4567,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,54.596,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gdpvalaa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",962,54.5684,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",978,59.3023,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",982,55.6061,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",982,55.578,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1140,67.8647,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1138,63.4848,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gdpvalaa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1138,63.4528,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaa-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1135,67.6004,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaa-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1139,63.5354,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gdpvalaa-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1139,63.5033,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1049,63.055,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1052,59.1414,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1052,59.1116,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaa-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1273,74.8943,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaa-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1271,70.202,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gdpvalaa-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1271,70.1666,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",936,57.0825,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",943,53.6364,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gdpvalaa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",943,53.6093,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaa-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1017,61.3636,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaa-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1017,57.3737,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gdpvalaa-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",1017,57.3448,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaa-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",554,36.8922,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaa-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",564,34.4949,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gdpvalaa-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",564,34.4775,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaa-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",554,36.8922,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaa-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",564,34.4949,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gdpvalaa-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","2026",564,34.4775,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-904","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",62.98,62.98,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-904--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",62.98,62.98,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:claude-fable-5:gdpval-aa:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",61.1495,61.1495,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1187","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",20.3305,20.3305,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1398","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",56.5145,56.5145,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1385","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",15.2015,15.2015,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-1018","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.995,49.995,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1018--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.995,49.995,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:claude-opus-4-7-adaptive:gdpval-aa:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.142,49.142,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["evidence-2026-07-1170","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.006,55.006,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1170--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.006,55.006,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:claude-opus-4-8:gdpval-aa:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",53.894000000000005,53.894000000000005,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:claude-opus-5:gdpval-aa:2026-08-29","claude-opus-5","Claude Opus 5","Claude Opus 5 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",66.197,66.197,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-5-2026-08-29","aa-current-claude-opus-5-2026-08-29","Claude Opus 5 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:claude-opus-5:gdpval-aa:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",66.102,66.102,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-2055","claude-opus-5","Claude Opus 5","GDPval-AA v2 Elo as published in the Anthropic Claude Opus 5 launch table.",null,"GDPval-AA v2 Elo as published in the Anthropic Claude Opus 5 launch table.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",1861,100,"elo","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports GDPval-AA v2 Elo 1861. Ranking continues to use independent AA percent rows when available; Elo retained as reference-only."],["evidence-2026-07-1768","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",35.871,35.871,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1716","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",18.0275,18.0275,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1370","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",27.623,27.623,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1008","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.858,43.858,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1008--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.858,43.858,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-956","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.336,55.336,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-956--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.336,55.336,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:claude-sonnet-5:gdpval-aa:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.340999999999994,54.340999999999994,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-2032","claude-sonnet-5","Claude Sonnet 5","GDPval-AA v2 Elo 1607 as published in Gemini 3.6 evaluation table.",null,"GDPval-AA v2 Elo 1607 as published in Gemini 3.6 evaluation table.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.35,55.35,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1607; normalized with (Elo-500)/2000 × 100."],["evidence-2026-07-1624","command-a-plus","Command A+","Command A+",null,"Command A+","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",10.71,10.71,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1530","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",19.464,19.464,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1515","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",18.274,18.274,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1154","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",34.427,34.427,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1154--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",34.427,34.427,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:deepseek-v4-flash-vision-exp:gdpval-aa:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",58.1405,58.1405,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual gdpval-aa result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["evidence-2026-07-1097","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",40.3605,40.3605,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1097--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",40.3605,40.3605,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:deepseek-v4-pro-0813:gdpval-aa:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",53.886,53.886,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["evidence-2026-07-1790","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",8.2695,8.2695,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-2053","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","GDPval-AA v2 Elo 642 as published for 3.1 Flash-Lite comparison.",null,"GDPval-AA v2 Elo 642 as published for 3.1 Flash-Lite comparison.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",7.1,7.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 642; normalized with (Elo-500)/2000 × 100."],["evidence-2026-07-1069","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",7.0785,7.0785,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1212","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",23.1155,23.1155,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-2011","gemini-3-5-flash","Gemini 3.5 Flash","GDPval-AA v2 Elo 1349 as published in Gemini 3.6 evaluation table for 3.5 Flash.",null,"GDPval-AA v2 Elo 1349 as published in Gemini 3.6 evaluation table for 3.5 Flash.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",42.45,42.45,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1349; normalized with (Elo-500)/2000 × 100."],["evidence-2026-07-913","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",42.442,42.442,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-913--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",42.442,42.442,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-2048","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis GDPval-AA v2 normalized score.",null,"Artificial Analysis GDPval-AA v2 normalized score.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",32,32,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA GDPval-AA normalized 32.0%."],["evidence-2026-07-2036","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDPval-AA v2 Elo 1140 as published in Google launch blog.",null,"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDPval-AA v2 Elo 1140 as published in Google launch blog.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",32,32,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1140; normalized with (Elo-500)/2000 × 100."],["evidence-2026-07-2008","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis GDPval-AA v2 normalized score as published on AA model page.",null,"Artificial Analysis GDPval-AA v2 normalized score as published on AA model page.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",46.1,46.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA GDPval-AA normalized 46.1% (Elo 1421)."],["aa-individual:gemini-3-6-flash:gdpval-aa:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",45.888,45.888,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1993","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDPval-AA v2 Elo as published by Google (sourced from AA leaderboard in methodology).",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDPval-AA v2 Elo as published by Google (sourced from AA leaderboard in methodology).","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",46.05,46.05,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1421; normalized with (Elo-500)/2000 × 100."],["aa-individual:gemini-3-7-flash:gdpval-aa:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.5595,49.5595,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:gdpval-aa:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1610","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",13.0665,13.0665,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["aa-current:gemma-4-26b-a4b:gdpval-aa:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",13.462,13.462,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1223","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",15.2015,15.2015,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1731","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",21.6785,21.6785,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1544","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",33.265,33.265,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1084","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",37.834,37.834,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1249","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",50.6945,50.6945,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1249--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",50.6945,50.6945,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:glm-5-2:gdpval-aa:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.8775,49.8775,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:gdpval-aa:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",63.1485,63.1485,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:gdpval-aa:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",63.1915,63.1915,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-flash-2026-08-27","aa-parity-model-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-979","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",4.5075,4.5075,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:gpt-4-1-mini:gdpval-aa:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0.342,0.342,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:gdpval-aa:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1330","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",28.74,28.74,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1330--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",28.74,28.74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1987","gpt-5-mini","GPT-5 mini","Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",21.8,21.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-gdpval-aa-normalized-2026-07-20","benchlm-gdpval-aa-normalized-2026-07-20","GDPval-AA Normalized Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/gdpvalAaNormalized","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public gdpval-aa leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1062","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",24.3725,24.3725,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1062--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",24.3725,24.3725,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:gpt-5-4:gdpval-aa:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",44.404999999999994,44.404999999999994,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1045","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",44.734,44.734,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1045--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",44.734,44.734,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1280","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",33.54,33.54,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1280--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",33.54,33.54,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1143","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",29.9945,29.9945,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1143--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",29.9945,29.9945,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:gpt-5-5:gdpval-aa:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.242999999999995,49.242999999999995,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-968","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.686,49.686,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-968--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",49.686,49.686,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-2019","gpt-5-6-luna","GPT-5.6 Luna","GDPval-AA v2 Elo 1584 as published in Gemini 3.6 evaluation table.",null,"GDPval-AA v2 Elo 1584 as published in Gemini 3.6 evaluation table.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.2,54.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1584; normalized with (Elo-500)/2000 × 100."],["aa-individual:gpt-5-6-luna:gdpval-aa:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",53.68300000000001,53.68300000000001,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-934","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.5905,54.5905,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-934--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.5905,54.5905,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:gpt-5-6-sol:gdpval-aa:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",60.571000000000005,60.571000000000005,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-941","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",62.391,62.391,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-941--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",62.391,62.391,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:gpt-5-6-terra:gdpval-aa:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",53.277499999999996,53.277499999999996,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-990","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.6475,54.6475,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-990--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",54.6475,54.6475,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1108","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",29.245,29.245,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-2025","grok-4-5","Grok 4.5","GDPval-AA v2 Elo 1535 as published in Gemini 3.6 evaluation table.",null,"GDPval-AA v2 Elo 1535 as published in Gemini 3.6 evaluation table.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",51.75,51.75,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Raw Elo 1535; normalized with (Elo-500)/2000 × 100."],["aa-individual:grok-4-5:gdpval-aa:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",50.907000000000004,50.907000000000004,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1136","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",51.7315,51.7315,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:grok-4-6:gdpval-aa:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",61.444500000000005,61.444500000000005,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1885","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",2.7365,2.7365,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1986","hy3","Hy3","Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",35.7,35.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-gdpval-aa-normalized-2026-07-20","benchlm-gdpval-aa-normalized-2026-07-20","GDPval-AA Normalized Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/gdpvalAaNormalized","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public gdpval-aa leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1179","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",25.4265,25.4265,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:kimi-k2-5-reasoning:gdpval-aa:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",25.2845,25.2845,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1408","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",34.5195,34.5195,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1426","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",34.3395,34.3395,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["aa-individual:kimi-k3:gdpval-aa:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",58.501999999999995,58.501999999999995,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1945","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis GDPval-AA v2 normalized score.",null,"Kimi K3; Artificial Analysis GDPval-AA v2 normalized score.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",59.2,59.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis / BenchLM-mirrored GDPval-AA normalized score 59.2 (Elo ~1685–1687 on public AA board)."],["evidence-2026-07-1780","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["aa-current:llama-4-maverick:gdpval-aa:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:gdpval-aa:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:gdpval-aa:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",26.6095,26.6095,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["evidence-2026-07-1580","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",40.3605,40.3605,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1164","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1567","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",32.2505,32.2505,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-924","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",38.27,38.27,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1053","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",32.8955,32.8955,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-999","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",44.7415,44.7415,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1038","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",6.6255,6.6255,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1239","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",21.4475,21.4475,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:mistral-medium-3-5-128b:gdpval-aa:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",21.7825,21.7825,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-949","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",4.3995,4.3995,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:mistral-small-4-reasoning:gdpval-aa:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",4.498,4.498,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-08-muse-glimmer-30b-gdpval-aa-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",953,22.65,"Elo","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Raw source value retained as 953 Elo. Only the documented registry normalization enters scoring."],["evidence-2026-08-muse-glimmer-30b-gdpval-aa-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",953,22.65,"Elo","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Raw source value retained as 953 Elo. Only the documented registry normalization enters scoring."],["evidence-2026-07-1260","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",32.1845,32.1845,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-individual:muse-spark-1-1:gdpval-aa:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.6595,43.6595,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1233","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.7165,43.7165,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1233--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.7165,43.7165,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1921","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.7,43.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1921--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",43.7,43.7,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["aa-individual:muse-spark-1-2:gdpval-aa:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",55.621,55.621,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual gdpval-aa result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["evidence-2026-08-muse-spark-1-2-gdpval-aa-v2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)",null,"Muse Spark 1.2 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",56.55,56.55,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-05","2026-08-05","production::aa-gdpval-muse-spark-1-2-2026-08-05","aa-gdpval-muse-spark-1-2-2026-08-05","GDPval-AA v2 leaderboard — Muse Spark 1.2","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-08-05","2026-08-05","2026-08-05","independently-verified","Artificial Analysis reports 1631 Elo with an 80% interval of -25/+18. Lumina stores the benchmark's published normalized form on a 0–100 scale."],["evidence-2026-08-muse-spark-1-2-gdpval-aa-v2--configuration--muse-spark-1-2-xhigh","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",56.55,56.55,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::aa-gdpval-muse-spark-1-2-2026-08-05","aa-gdpval-muse-spark-1-2-2026-08-05","GDPval-AA v2 leaderboard — Muse Spark 1.2","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-08-05","2026-08-05","2026-08-05","independently-verified","Artificial Analysis reports 1631 Elo with an 80% interval of -25/+18. Lumina stores the benchmark's published normalized form on a 0–100 scale."],["aa-current:nemotron-3-nano-30b:gdpval-aa:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1817","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",9.6625,9.6625,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1595","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",33.19,33.19,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1637","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",8.811,8.811,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1491","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",23.901,23.901,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1473","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",23.0955,23.0955,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["aa-current:qwen3-5-397b-reasoning:gdpval-aa:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",23.318,23.318,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:gdpval-aa:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",24.356,24.356,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1451","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",31.986,31.986,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1267","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",31.7695,31.7695,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:qwen3-6-35b-a3b:gdpval-aa:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",27.818,27.818,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1119","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",38.6575,38.6575,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["evidence-2026-07-1197","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",21.813,21.813,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gdpval-aa","aa-gdpval-aa","GDPval-AA v2 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/gdpval-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","AA GDPval-AA v2 normalized score (Elo mapped to 0–1 by Artificial Analysis)."],["aa-current:qwen-3-8-flash-next:gdpval-aa:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",62.139,62.139,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:gdpval-aa:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","gdpval-aa","GDPval-AA v2","agents","Artificial Analysis / OpenAI","v2",3.2695,3.2695,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-08-15-claude-fable-5-gdpval-aa-v2-elo-v2-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1743,62.15,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["xai-grok-4-6-release-gdpval-aa-v2-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1741,62.05,"Elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-08-15-claude-opus-4-8-gdpval-aa-v2-elo-v2-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1588,54.4,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-command-a-plus-gdpval-aa-v2-elo-v2-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",712,10.6,"Elo","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-gdpval-aa-v2-elo-v2-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1187,34.35,"Elo","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-pro-0813-gdpval-aa-v2-elo-v2-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1590,54.5,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-glm-5-2-gdpval-aa-v2-elo-v2-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1508,50.4,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-glm-5-3-gdpval-aa-v2-elo-v2","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1769,63.45,"Elo","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-gpt-5-6-sol-gdpval-aa-v2-elo-v2-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1730,61.5,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["xai-grok-4-6-release-gdpval-aa-v2-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1728,61.4,"Elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-gdpval-aa-v2-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1526,51.3,"Elo","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-gdpval-aa-v2-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1753,62.65,"Elo","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["evidence-2026-08-15-kimi-k3-gdpval-aa-v2-elo-v2-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1682,59.1,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-mimo-v2-5-gdpval-aa-v2-elo-v2-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1145,32.25,"Elo","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-gdpval-aa-v2-elo-v2-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",929,21.45,"Elo","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["nvidia-nemotron-3-5-lightning-gdpval-aa-v2-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",832,16.6,"Elo","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-08-15-qwen-3-8-max-gdpval-aa-v2-elo-v2-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1739,61.95,"Elo","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote: models are evaluated by Artificial Analysis. Stored as Elo, not as an AA Intelligence Index."],["evidence-2026-08-15-solar-open2-250b-gdpval-aa-v2-elo-v2","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2",1128,31.4,"Elo","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-claude-opus-5","claude-opus-5","Claude Opus 5","Claude Opus 5 (reasoning configuration not stated in chart)","claude-opus-5-meta-muse-12-release-unspecified","Claude Opus 5 (reasoning configuration not stated in chart)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1852,67.6,"Elo","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["google-gemini-37-eval-gdpval-aa-v2-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1598,54.9,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdpval-aa-v2-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1422,46.1,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-gemini-3-6-flash","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (reasoning configuration not stated in chart)","gemini-3-6-flash-meta-muse-12-release-unspecified","Gemini 3.6 Flash (reasoning configuration not stated in chart)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1423,46.15,"Elo","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["google-gemini-37-eval-gdpval-aa-v2-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1525,51.25,"Elo","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-gdpval-aa-v2-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1578,53.9,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-gpt-5-6-terra","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (reasoning configuration not stated in chart)","gpt-5-6-terra-meta-muse-12-release-unspecified","GPT-5.6 Terra (reasoning configuration not stated in chart)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1577,53.85,"Elo","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-grok-4-5","grok-4-5","Grok 4.5","Grok 4.5 (reasoning configuration not stated in chart)","grok-4-5-meta-muse-12-release-unspecified","Grok 4.5 (reasoning configuration not stated in chart)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1526,51.3,"Elo","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-muse-spark-1-1","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (reasoning configuration not stated in chart)","muse-spark-1-1-meta-muse-12-release-unspecified","Muse Spark 1.1 (reasoning configuration not stated in chart)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1371,43.55,"Elo","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["google-gemini-37-eval-gdpval-aa-v2-muse-spark-1-2-2026-08-13","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2","muse-spark-1-2-xhigh","Muse Spark 1.2","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1628,56.4,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-gdpval-aa-v2-muse-spark-1-2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","gdpval-aa-v2-elo","GDPval-AA v2 (Elo)","agents","Artificial Analysis / OpenAI","v2 Elo",1631,56.55,"Elo","higher","2.0.0","reference-only","direct","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-gdpval-chart","refresh-meta-muse-spark-1-2-gdpval-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/gdpval-aa-v2-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Exact Muse Spark 1.2 result retained with the published xhigh setting."],["benchlm-ref-claude-4-sonnet-gertlabs-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.66,29.6069,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-gertlabs-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.66,29.6069,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-gertlabs-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.66,29.6069,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gertlabs-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.23,81.53,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gertlabs-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.23,81.53,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gertlabs-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.23,81.53,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gertlabs-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gertlabs-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gertlabs-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-gertlabs-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",65.59,84.4041,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-gertlabs-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",65.59,84.4041,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-gertlabs-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",65.59,84.4041,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gertlabs-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.97,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gertlabs-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.97,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gertlabs-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.97,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gertlabs-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",48.51,48.3094,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gertlabs-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",48.51,48.3094,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gertlabs-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",48.51,48.3094,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gertlabs-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.92,78.7616,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gertlabs-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.92,78.7616,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gertlabs-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.92,78.7616,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-gertlabs-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.57,8.284,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-gertlabs-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.57,8.284,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-gertlabs-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.57,8.284,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gertlabs-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.35,60.6509,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gertlabs-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.35,60.6509,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gertlabs-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.35,60.6509,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gertlabs-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.28,52.0499,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gertlabs-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.28,52.0499,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gertlabs-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.28,52.0499,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gertlabs-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.01,34.5731,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gertlabs-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.01,34.5731,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gertlabs-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.01,34.5731,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-gertlabs-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.63,65.4691,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-gertlabs-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.63,65.4691,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-gertlabs-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.63,65.4691,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-gertlabs-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",63.23,79.4167,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-gertlabs-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",63.23,79.4167,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-gertlabs-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",63.23,79.4167,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.46,27.071,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.46,27.071,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.46,27.071,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gertlabs-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.87,65.9763,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gertlabs-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.87,65.9763,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gertlabs-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.87,65.9763,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gertlabs-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gertlabs-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gertlabs-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",61.85,76.5004,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gertlabs-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",35.26,20.3085,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gertlabs-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",35.26,20.3085,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gertlabs-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",35.26,20.3085,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gertlabs-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.95,30.2198,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gertlabs-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.95,30.2198,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gertlabs-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.95,30.2198,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gertlabs-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.99,53.5503,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gertlabs-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.99,53.5503,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gertlabs-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.99,53.5503,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gertlabs-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",60.11,72.8233,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gertlabs-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",60.11,72.8233,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gertlabs-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",60.11,72.8233,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-gertlabs-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",30.76,10.7988,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-gertlabs-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",30.76,10.7988,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-gertlabs-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",30.76,10.7988,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gertlabs-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",25.65,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gertlabs-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",25.65,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gertlabs-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",25.65,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gertlabs-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",41.24,32.9459,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gertlabs-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",41.24,32.9459,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-gertlabs-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",41.24,32.9459,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-gertlabs-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.68,50.7819,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-gertlabs-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.68,50.7819,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-gertlabs-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.68,50.7819,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gertlabs-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.54,44.1462,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gertlabs-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.54,44.1462,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gertlabs-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.54,44.1462,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-gertlabs-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.79,55.2409,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-gertlabs-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.79,55.2409,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-gertlabs-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.79,55.2409,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-gertlabs-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",57.47,67.2443,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-gertlabs-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",57.47,67.2443,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-gertlabs-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",57.47,67.2443,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gertlabs-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.89,82.9248,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gertlabs-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.89,82.9248,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gertlabs-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.89,82.9248,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gertlabs-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.93,99.9155,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gertlabs-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.93,99.9155,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gertlabs-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",72.93,99.9155,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gertlabs-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.61,8.3686,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gertlabs-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.61,8.3686,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-gertlabs-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",29.61,8.3686,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-gertlabs-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.34,35.2705,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-gertlabs-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.34,35.2705,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-gertlabs-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.34,35.2705,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-gertlabs-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",47.32,45.7946,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-gertlabs-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",47.32,45.7946,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-gertlabs-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",47.32,45.7946,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gertlabs-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.36,26.8597,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gertlabs-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.36,26.8597,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gertlabs-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",38.36,26.8597,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gertlabs-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.86,38.4827,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gertlabs-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.86,38.4827,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gertlabs-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.86,38.4827,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-build-0-1-gertlabs-2026-07-21","grok-build-0-1","Grok Build 0.1","Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.15,49.6619,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-build-0-1-gertlabs-2026-07-27","grok-build-0-1","Grok Build 0.1","Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.15,49.6619,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-build-0-1-gertlabs-2026-08-01","grok-build-0-1","Grok Build 0.1","Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",49.15,49.6619,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gertlabs-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.91,23.7954,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gertlabs-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.91,23.7954,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gertlabs-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.91,23.7954,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gertlabs-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.58,14.645,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gertlabs-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.58,14.645,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gertlabs-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.58,14.645,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gertlabs-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",45.88,42.7515,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gertlabs-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",45.88,42.7515,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gertlabs-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",45.88,42.7515,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gertlabs-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.82,65.8707,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gertlabs-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.82,65.8707,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gertlabs-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",56.82,65.8707,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-gertlabs-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.68,23.3094,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-gertlabs-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.68,23.3094,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-gertlabs-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",36.68,23.3094,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-gertlabs-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.89,44.8859,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-gertlabs-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.89,44.8859,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-gertlabs-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.89,44.8859,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gertlabs-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.7,78.2967,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gertlabs-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.7,78.2967,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-gertlabs-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",62.7,78.2967,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gertlabs-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",40.4,31.1708,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gertlabs-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",40.4,31.1708,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gertlabs-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",40.4,31.1708,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.1,28.4235,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.1,28.4235,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.1,28.4235,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-gertlabs-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.74,38.2291,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-gertlabs-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.74,38.2291,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-gertlabs-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",43.74,38.2291,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gertlabs-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.41,29.0786,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gertlabs-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.41,29.0786,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gertlabs-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",39.41,29.0786,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gertlabs-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.76,44.6112,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gertlabs-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.76,44.6112,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gertlabs-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",46.76,44.6112,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",28.96,6.9949,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",28.96,6.9949,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",28.96,6.9949,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gertlabs-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.84,61.6864,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gertlabs-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.84,61.6864,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gertlabs-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",54.84,61.6864,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gertlabs-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.6,52.7261,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gertlabs-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.6,52.7261,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gertlabs-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",50.6,52.7261,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.65,35.9256,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.65,35.9256,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",42.65,35.9256,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gertlabs-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.27,81.6145,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gertlabs-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.27,81.6145,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gertlabs-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",64.27,81.6145,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gertlabs-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.57,54.776,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gertlabs-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.57,54.776,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-gertlabs-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",51.57,54.776,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gertlabs-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.55,14.5816,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gertlabs-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.55,14.5816,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gertlabs-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gertlabs","Gert Labs Composite Game Benchmark","agents","Gert Labs","2026",32.55,14.5816,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-harvey-lab-aa-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey","August 2026",90.1,90.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-harvey-lab-aa-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey","August 2026",85.1,85.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-harvey-lab-aa-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey","August 2026",90.7,90.7,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-harvey-lab-aa-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey","August 2026",85.2,85.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-07-911","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,14.1667,14.1667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-911--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,14.1667,14.1667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["xai-grok-4-6-release-harvey-lab-vals-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,11.3,11.3,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-1195","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1390","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-1177","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,7.5,7.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1177--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,7.5,7.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-2066","claude-opus-5","Claude Opus 5","Legal Agent Benchmark held-out split as published in the Anthropic Claude Opus 5 launch table.",null,"Legal Agent Benchmark held-out split as published in the Anthropic Claude Opus 5 launch table.","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,11.7,11.7,"percent","higher","1.6.0","ranking-eligible","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 11.7% on Legal Agent Benchmark (held-out). Mapped to harvey-lab-aa; Fable 5 is 13.3% in the same table vs ~14.2% on existing AA rows."],["evidence-2026-07-1017","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,4.1667,4.1667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1017--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,4.1667,4.1667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-961","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,5,5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-961--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,5,5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1162","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,1.6667,1.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1162--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,1.6667,1.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1106","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,3.3333,3.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1106--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,3.3333,3.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1077","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1221","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-922","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,1.6667,1.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-922--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,1.6667,1.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1231","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1258","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,7.5,7.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1258--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,7.5,7.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1290","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1290--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1152","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1152--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-977","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,4.1667,4.1667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-977--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,4.1667,4.1667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-940","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,5,5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-940--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,5,5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["xai-grok-4-6-release-harvey-lab-vals-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,2.5,2.5,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-998","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,2.5,2.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-998--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,2.5,2.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1117","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1141","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,13.3333,13.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["xai-grok-4-6-release-harvey-lab-vals-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,12.9,12.9,"percent","higher","1.8.0","ranking-eligible","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-harvey-lab-vals-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,15.8,15.8,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["evidence-2026-07-1418","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1434","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0.8333,0.8333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1953","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis Harvey LAB-AA evaluation.",null,"Kimi K3; Artificial Analysis Harvey LAB-AA evaluation.","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,26.7,26.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis Harvey LAB-AA score for Kimi K3."],["evidence-2026-07-1586","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,3.3333,3.3333,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-933","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1006","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,6.6667,6.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1246","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0.8333,0.8333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1238","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,8.3333,8.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1238--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,8.3333,8.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1923","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,8.3,8.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1923--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,8.3,8.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1603","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,3.3333,3.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1483","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1127","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["evidence-2026-07-1205","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","harvey-lab-aa","Harvey LAB-AA","agents","Artificial Analysis / Harvey",null,1.6667,1.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-harvey-lab","aa-harvey-lab","Harvey LAB-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/harvey-lab-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Harvey LAB-AA all-pass rate from Artificial Analysis."],["benchlm-ref-agents-a1-hlewithtools-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.6,51,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-hlewithtools-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.6,37.3626,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-hlewithtools-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.6,37.3626,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hlewithtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",64.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hlewithtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",64.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2060","claude-opus-5","Claude Opus 5","Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.",null,"Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",64.7,64.7,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Mirror of the with-tools HLE table score on the BenchLM-named family for directory completeness."],["benchlm-ref-claude-sonnet-5-hlewithtools-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",57.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hlewithtools-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",57.4,73.2601,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hlewithtools-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",57.4,73.2601,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,14.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,14.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,10.6227,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,10.6227,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,10.6227,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hlewithtools-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",40.3,10.6227,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,38.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,28.2051,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,28.2051,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,38.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,28.2051,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",45.1,28.2051,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,36.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,36.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,26.7399,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,26.7399,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,26.7399,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hlewithtools-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",44.7,26.7399,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,54,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,39.5604,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,39.5604,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,54,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,39.5604,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",48.2,39.5604,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlewithtools-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",37.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlewithtools-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",37.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlewithtools-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",37.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hlewithtools-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",53.5,80.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hlewithtools-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",53.5,58.9744,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hlewithtools-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",53.5,58.9744,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-hlewithtools-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.2,49,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-hlewithtools-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.2,35.8974,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-hlewithtools-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","2026",47.2,35.8974,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-hle-with-tools-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",63,63,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",57.9,57.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",45.1,45.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",51.5,51.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",48.2,48.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",60,60,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",54.7,54.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-with-tools-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","August 2026, with tools",56,56,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-08-15-claude-fable-5-benchlm-hlewithtools-with-tools-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",63.9,63.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-claude-opus-4-8-benchlm-hlewithtools-with-tools-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",57.9,57.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-deepseek-v4-pro-0813-benchlm-hlewithtools-with-tools-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",60,60,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-glm-5-2-benchlm-hlewithtools-with-tools-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",54.7,54.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-glm-5-3-benchlm-hlewithtools-with-tools","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",62.5,62.5,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-gpt-5-6-sol-benchlm-hlewithtools-with-tools-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",64.5,64.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-kimi-k3-benchlm-hlewithtools-with-tools-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",59.8,59.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-08-15-qwen-3-8-max-benchlm-hlewithtools-with-tools-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","benchlm-hlewithtools","Humanity's Last Exam with tools","agents","DeepSeek-AI","with tools",56.2,56.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=0.95, max generation 163840, context 300000."],["evidence-2026-07-1194","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,27.307,27.307,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1025","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,46.6573,46.6573,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1025--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,46.6573,46.6573,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1016","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,39.8023,39.8023,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1016--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,39.8023,39.8023,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1161","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,31.516,31.516,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1161--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,31.516,31.516,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1105","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,38.3239,38.3239,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1105--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,38.3239,38.3239,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1220","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,30.3309,30.3309,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-921","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,40.3484,40.3484,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-921--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,40.3484,40.3484,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1618","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,23.6347,23.6347,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1230","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,37.2881,37.2881,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1091","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,40.2542,40.2542,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1257","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,42.6554,42.6554,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1257--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,42.6554,42.6554,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1289","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,35.2166,35.2166,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1289--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,35.2166,35.2166,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1151","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,24.3879,24.3879,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1151--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,24.3879,24.3879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-976","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,45.8098,45.8098,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-976--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,45.8098,45.8098,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-939","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,40.3202,40.3202,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-939--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,40.3202,40.3202,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-948","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,56.2147,56.2147,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-948--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,56.2147,56.2147,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-997","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,51.0358,51.0358,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-997--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,51.0358,51.0358,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1116","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,32.7213,32.7213,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1417","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,31.1864,31.1864,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-932","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,38.2298,38.2298,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1061","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,26.4595,26.4595,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["evidence-2026-07-1826","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,1.1299,1.1299,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1509","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,35.4991,35.4991,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1710","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,21.516,21.516,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1482","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,34.0866,34.0866,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1126","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","itbench-aa","ITBench-AA","agents","Artificial Analysis / IBM",null,42.467,42.467,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-itbench","aa-itbench","ITBench-AA Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/itbench-aa","2026-07-15","2026-07-15","2026-07-15","source-checked","ITBench-AA SRE score from Artificial Analysis."],["benchlm-ref-claude-4-sonnet-jobbench-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.4,21.3775,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-jobbench-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.4,21.3775,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-jobbench-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.4,21.3775,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-jobbench-2026-07-21","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",21.9,28.9582,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-jobbench-2026-07-27","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",21.9,28.9582,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-jobbench-2026-08-01","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",21.9,28.9582,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-jobbench-2026-07-21","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",16,16.1793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-jobbench-2026-07-27","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",16,16.1793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-jobbench-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",16,16.1793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-jobbench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",32.3,51.4836,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-jobbench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",32.3,51.4836,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-jobbench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",32.3,51.4836,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:jobbench:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.6,36.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["benchlm-ref-claude-opus-4-6-jobbench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.7,61.0136,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-jobbench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.7,61.0136,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-jobbench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.7,61.0136,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-jobbench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",45.9,80.94,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-jobbench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",45.9,80.94,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-jobbench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",45.9,80.94,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-jobbench-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",27.7,41.5205,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-jobbench-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",27.7,41.5205,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-jobbench-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",27.7,41.5205,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-jobbench-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.9,61.4468,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-jobbench-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.9,61.4468,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-jobbench-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",36.9,61.4468,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:jobbench:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",41.3,41.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["benchlm-ref-gemini-3-flash-jobbench-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-jobbench-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-jobbench-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-jobbench-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-jobbench-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-jobbench-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",11.4,6.2162,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-jobbench-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.53,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-jobbench-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26.2,38.2716,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-jobbench-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26.2,38.2716,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-jobbench-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26.2,38.2716,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-jobbench-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",34.3,55.8155,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-jobbench-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",34.3,55.8155,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-jobbench-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",34.3,55.8155,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-jobbench-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26,37.8384,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-jobbench-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26,37.8384,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-jobbench-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",26,37.8384,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-jobbench-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",33.7,54.5159,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-jobbench-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",33.7,54.5159,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-jobbench-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",33.7,54.5159,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-jobbench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",38.9,65.7786,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-jobbench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",38.9,65.7786,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-jobbench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",38.9,65.7786,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-jobbench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",42.7,74.0091,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-jobbench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",42.7,74.0091,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-jobbench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",42.7,74.0091,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-jobbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.73,0.4332,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-jobbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.73,0.4332,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-jobbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",8.73,0.4332,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-jobbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",52.9,96.1014,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-jobbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",52.9,96.1014,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-jobbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",52.9,96.1014,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-jobbench-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",54.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-jobbench-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",54.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-jobbench-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",54.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-jobbench-2026-07-21","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.5,21.5941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-jobbench-2026-07-27","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.5,21.5941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-jobbench-2026-08-01","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",18.5,21.5941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-jobbench-2026-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",21.8,21.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:jobbench:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",27.6,27.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-jobbench-2026-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",27.6,27.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-qwen-3-8-max-jobbench","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",53.4,53.4,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 53.4. Stored as provider-reported reference evidence under the existing JobBench registry entry."],["evidence-2026-08-qwen-3-8-max-jobbench--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",53.4,53.4,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 53.4. Stored as provider-reported reference evidence under the existing JobBench registry entry."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:jobbench:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",33.4,33.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-jobbench-2026","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",33.4,33.4,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-jobbench:cell:language:jobbench:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-jobbench","JobBench","agents","Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","2026",55.7,55.7,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-kimiclaw247","Kimi Claw 24/7 Bench","agents","Moonshot AI","2026",46.9,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-kimiclaw247","Kimi Claw 24/7 Bench","agents","Moonshot AI","2026",46.9,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-kimiclaw247","Kimi Claw 24/7 Bench","agents","Moonshot AI","2026",46.9,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchallpass-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchallpass","Legal Agent Benchmark all-pass rate — Anthropic harness","agents","Harvey AI and Anthropic","2026",23.58,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchallpass-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchallpass","Legal Agent Benchmark all-pass rate — Anthropic harness","agents","Harvey AI and Anthropic","2026",23.58,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchheldoutallpass-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchheldoutallpass","Legal Agent Benchmark all-pass rate — Harvey held-out set","agents","Harvey AI","2026",11.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchheldoutallpass-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchheldoutallpass","Legal Agent Benchmark all-pass rate — Harvey held-out set","agents","Harvey AI","2026",11.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchcriterionpass-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","agents","Harvey AI and Anthropic","2026",93.74,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchcriterionpass-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","agents","Harvey AI and Anthropic","2026",93.74,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchheldoutcriterionpass-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchheldoutcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","agents","Harvey AI","2026",94.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-legalagentbenchheldoutcriterionpass-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-legalagentbenchheldoutcriterionpass","Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","agents","Harvey AI","2026",94.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-362","claude-fable-5","Claude Fable 5","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",62.9,62.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-757","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",40.9,40.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-747","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",48.5,48.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-745","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",54.5,54.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-364","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",58.3,58.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-748","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",46.2,46.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-764","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",31.8,31.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-756","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",41.7,41.7,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-375","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",24.2,24.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-367","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",53.8,53.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-372","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",40.9,40.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-760","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",36.4,36.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-753","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",43.2,43.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-762","glm-5-turbo","GLM-5-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",33.3,33.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-752","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",43.2,43.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-746","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",50.8,50.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-763","glm-5v-turbo","GLM-5V-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",32.6,32.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-749","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",45.5,45.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-368","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",53,53,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-365","gpt-5-4","GPT-5.4","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",57.6,57.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-755","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",42.4,42.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-363","gpt-5-5","GPT-5.5","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",60.6,60.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-361","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",65.9,65.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-366","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",57.6,57.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-373","grok-4-3","Grok 4.3","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",37.9,37.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-374","kimi-k2-5","Kimi K2.5","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",34.8,34.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-761","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",34.8,34.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-758","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",40.9,40.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-754","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",43.2,43.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-759","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",39.4,39.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-371","minimax-m3","MiniMax M3","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",42.4,42.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-377","mistral-large-3","Mistral Large 3","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",15.9,15.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-376","mistral-small-4","Mistral Small 4","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",17.4,17.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-750","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",45.5,45.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-751","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",43.9,43.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-369","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",50.8,50.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["evidence-2026-07-370","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).",null,"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","long-horizon-terminal-bench","Long-Horizon-Terminal-Bench","agents","Long-Horizon-Terminal-Bench authors","2026-07",47,47,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM Terminal-Bench Hard public table into long-horizon-terminal-bench registry slot."],["benchlm-ref-claude-opus-4-5-mcpatlas-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",42.3,21.843,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mcpatlas-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",42.3,21.843,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mcpatlas-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",42.3,21.843,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mcpatlas-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",77.3,81.57,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mcpatlas-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",77.3,81.57,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mcpatlas-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",77.3,81.57,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-mcpatlas-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",82.2,89.9317,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-mcpatlas-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",82.2,89.9317,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-mcpatlas-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",82.2,89.9317,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-mcpatlas-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",85.8,96.0751,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-mcpatlas-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",85.8,96.0751,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mcpatlas-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",67.4,64.6758,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mcpatlas-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",64,58.8737,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mcpatlas-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",64,58.8737,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mcpatlas-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",64,58.8737,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69,67.4061,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mcpatlas-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mcpatlas-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69.4,68.0887,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mcpatlas-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69.4,68.0887,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mcpatlas-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",69.4,68.0887,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.6,75.256,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mcpatlas-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",83.6,92.3208,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mcpatlas-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",83.6,92.3208,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mcpatlas-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",83.6,92.3208,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcpatlas-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",31.1,2.7304,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcpatlas-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",31.1,2.7304,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcpatlas-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",31.1,2.7304,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mcpatlas-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",71.8,72.1843,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mcpatlas-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",71.8,72.1843,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mcpatlas-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",71.8,72.1843,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mcpatlas-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.8,80.7167,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mcpatlas-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.8,80.7167,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mcpatlas-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.8,80.7167,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mcpatlas-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",70.6,70.1365,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mcpatlas-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",70.6,70.1365,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mcpatlas-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",70.6,70.1365,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mcpatlas-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",57.7,48.1229,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mcpatlas-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",57.7,48.1229,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mcpatlas-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",57.7,48.1229,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mcpatlas-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",56.1,45.3925,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mcpatlas-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",56.1,45.3925,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mcpatlas-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",56.1,45.3925,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mcpatlas-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",75.3,78.157,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mcpatlas-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",75.3,78.157,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mcpatlas-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",75.3,78.157,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mcpatlas-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.1,76.1092,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mcpatlas-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.1,76.1092,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mcpatlas-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.1,76.1092,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-mcpatlas-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",79.6,85.4949,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcpatlas-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",29.5,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcpatlas-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",29.5,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcpatlas-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",29.5,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mcpatlas-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",55.9,45.0512,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mcpatlas-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",55.9,45.0512,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mcpatlas-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",55.9,45.0512,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpatlas-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76,79.3515,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpatlas-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76,79.3515,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpatlas-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76,79.3515,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mcpatlas-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",84.2,93.3447,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mcpatlas-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",84.2,93.3447,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mcpatlas-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",84.2,93.3447,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mcpatlas-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mcpatlas-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mcpatlas-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",74.2,76.2799,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mcpatlas-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",88.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mcpatlas-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",88.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mcpatlas-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",88.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcpatlas-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",46.1,28.3276,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcpatlas-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",46.1,28.3276,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcpatlas-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",46.1,28.3276,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcpatlas-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",48.2,31.9113,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcpatlas-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",48.2,31.9113,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcpatlas-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",48.2,31.9113,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",62.8,56.8259,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",62.8,56.8259,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",62.8,56.8259,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mcpatlas-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.4,80.0341,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mcpatlas-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.4,80.0341,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mcpatlas-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",76.4,80.0341,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mcpatlas-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.2,74.5734,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mcpatlas-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.2,74.5734,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mcpatlas-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mcp-atlas","MCP Atlas","agents","Google","2026",73.2,74.5734,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-claude-fable-5","claude-fable-5","Claude Fable 5","Claude Fable 5 (reasoning configuration not stated in chart)","claude-fable-5-meta-muse-12-release-unspecified","Claude Fable 5 (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",83.3,83.3,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-652","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",42.3,42.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-901","claude-opus-4-5","Claude Opus 4.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",42.3,42.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-241","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",82.2,82.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-889","claude-opus-4-8","Claude Opus 4.8","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",82.2,82.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-claude-opus-5","claude-opus-5","Claude Opus 5","Claude Opus 5 (reasoning configuration not stated in chart)","claude-opus-5-meta-muse-12-release-unspecified","Claude Opus 5 (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",85.8,85.8,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-248","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",64,64,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-898","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",64,64,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-247","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",69.4,69.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-897","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",69.4,69.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-012","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mcp-atlas","MCP Atlas","agents","Google","May 2026",78.2,78.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-002","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mcp-atlas","MCP Atlas","agents","Google","May 2026",83.6,83.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-240","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",83.6,83.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-888","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",83.6,83.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-gemini-3-5-flash","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (reasoning configuration not stated in chart)","gemini-3-5-flash-meta-muse-12-release-unspecified","Gemini 3.5 Flash (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",83.6,83.6,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-653","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",31.1,31.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-902","glm-5","GLM-5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",31.1,31.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-649","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",71.8,71.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-895","glm-5-1","GLM-5.1","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",71.8,71.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-648","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",76.8,76.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-890","glm-5-2","GLM-5.2","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",76.8,76.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-246","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",70.6,70.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-896","gpt-5-4","GPT-5.4","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",70.6,70.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-650","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",56.1,56.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-899","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",56.1,56.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-021","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mcp-atlas","MCP Atlas","agents","Google","May 2026",75.3,75.3,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-243","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",75.3,75.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-892","gpt-5-5","GPT-5.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",75.3,75.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-gpt-5-6-sol","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (reasoning configuration not stated in chart)","gpt-5-6-sol-meta-muse-12-release-unspecified","GPT-5.6 Sol (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",81.8,81.8,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-249","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",29.5,29.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-903","kimi-k2-5","Kimi K2.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",29.5,29.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-kimi-k3","kimi-k3","Kimi K3","Kimi K3 (reasoning configuration not stated in chart)","kimi-k3-meta-muse-12-release-unspecified","Kimi K3 (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",82.3,82.3,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-1938","kimi-k3","Kimi K3","MCP Atlas 500-task public subset; 100-turn limit; Gemini 3.1 Pro judge; max reasoning.",null,"MCP Atlas 500-task public subset; 100-turn limit; Gemini 3.1 Pro judge; max reasoning.","mcp-atlas","MCP Atlas","agents","Google","May 2026",84.2,84.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 84.2 on the public MCP Atlas subset."],["evidence-2026-07-1938--configuration--kimi-k3-max","kimi-k3","Kimi K3","MCP Atlas 500-task public subset; 100-turn limit; Gemini 3.1 Pro judge; max reasoning.","kimi-k3-max","MCP Atlas 500-task public subset; 100-turn limit; Gemini 3.1 Pro judge; max reasoning.","mcp-atlas","MCP Atlas","agents","Google","May 2026",84.2,84.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 84.2 on the public MCP Atlas subset."],["evidence-2026-07-244","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",74.2,74.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-893","minimax-m3","MiniMax M3","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",74.2,74.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-647","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",88.1,88.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-887","muse-spark-1-1","Muse Spark 1.1","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",88.1,88.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-muse-spark-1-1","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (reasoning configuration not stated in chart)","muse-spark-1-1-meta-muse-12-release-unspecified","Muse Spark 1.1 (reasoning configuration not stated in chart)","mcp-atlas","MCP Atlas","agents","Google","May 2026",88.1,88.1,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-1904","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MCP Atlas).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MCP Atlas).","mcp-atlas","MCP Atlas","agents","Google","May 2026",88.1,88.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for MCP Atlas; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1904--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MCP Atlas).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MCP Atlas).","mcp-atlas","MCP Atlas","agents","Google","May 2026",88.1,88.1,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for MCP Atlas; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-08-15-meta-muse-12-mcp-atlas-muse-spark-1-2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","mcp-atlas","MCP Atlas","agents","Google","May 2026",90.3,90.3,"percent","higher","2.0.0","reference-only","direct","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-mcp-atlas-chart","refresh-meta-muse-spark-1-2-mcp-atlas-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/mcp-atlas-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Exact Muse Spark 1.2 result retained with the published xhigh setting."],["evidence-2026-08-muse-spark-1-2-mcp-atlas","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) in the Scale AI MCP Atlas harness",null,"Muse Spark 1.2 (xhigh) in the Scale AI MCP Atlas harness","mcp-atlas","MCP Atlas","agents","Google","May 2026",90.3,90.3,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports a 90.3% pass rate from Scale AI's evaluation of 1,000 tasks spanning 36 MCP servers and 220 tools."],["evidence-2026-08-muse-spark-1-2-mcp-atlas--configuration--muse-spark-1-2-xhigh","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) in the Scale AI MCP Atlas harness","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh) in the Scale AI MCP Atlas harness","mcp-atlas","MCP Atlas","agents","Google","May 2026",90.3,90.3,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports a 90.3% pass rate from Scale AI's evaluation of 1,000 tasks spanning 36 MCP servers and 220 tools."],["evidence-2026-07-651","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",48.2,48.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-900","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",48.2,48.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-242","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",76.4,76.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-891","qwen-3-7-max","Qwen3.7-Max","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",76.4,76.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-245","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",73.2,73.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-894","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","mcp-atlas","MCP Atlas","agents","Google","May 2026",73.2,73.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-muse-glimmer-30b-mcp-atlas-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","mcp-atlas","MCP Atlas","agents","Google","Public / 500 tasks",75.5,75.5,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported mean pass rate using a 0.75 judge threshold."],["evidence-2026-08-muse-glimmer-30b-mcp-atlas-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","mcp-atlas","MCP Atlas","agents","Google","Public / 500 tasks",75.5,75.5,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported mean pass rate using a 0.75 judge threshold."],["evidence-2026-08-15-command-a-plus-mcp-atlas-standard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","mcp-atlas","MCP Atlas","agents","Google","standard",27.2,27.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-mcp-atlas-standard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","mcp-atlas","MCP Atlas","agents","Google","standard",58.2,58.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-mcp-atlas-standard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","mcp-atlas","MCP Atlas","agents","Google","standard",63.9,63.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-mcp-atlas-standard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","mcp-atlas","MCP Atlas","agents","Google","standard",30.7,30.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-mcp-atlas-standard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","mcp-atlas","MCP Atlas","agents","Google","standard",34.4,34.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-mcp-atlas-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","mcp-atlas","MCP Atlas","agents","Google","standard",58.2,58.2,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-opus-5-mcpatlasclaimcoverage-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcpatlasclaimcoverage","MCP-Atlas mean claim coverage","agents","Anthropic","2026",89.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-mcpatlasclaimcoverage-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcpatlasclaimcoverage","MCP-Atlas mean claim coverage","agents","Anthropic","2026",89.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mcptasks-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",71.8,84.106,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mcptasks-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",71.8,84.106,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mcptasks-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",71.8,84.106,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcptasks-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",60.8,11.2583,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcptasks-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",60.8,11.2583,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mcptasks-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",60.8,11.2583,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcptasks-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",59.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcptasks-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",59.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mcptasks-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",59.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcptasks-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcptasks-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mcptasks-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcptasks-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.1,99.3377,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcptasks-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.1,99.3377,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mcptasks-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mcptasks","MCP-Tasks","agents","Qwen","2026",74.1,99.3377,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mcpmarkverified","MCPMark-Verified","agents","MCPMark","2026",81.1,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mcpmarkverified","MCPMark-Verified","agents","MCPMark","2026",81.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mcpmarkverified","MCPMark-Verified","agents","MCPMark","2026",81.1,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mlebenchlite-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mlebenchlite","MLE-Bench Lite","agents","MiniMax","2026",66.6,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mlebenchlite-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mlebenchlite","MLE-Bench Lite","agents","MiniMax","2026",66.6,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mlebenchlite-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mlebenchlite","MLE-Bench Lite","agents","MiniMax","2026",66.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmclawbench-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",23.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmclawbench-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",23.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmclawbench-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",23.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmclawbench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",62.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmclawbench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",62.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmclawbench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmclawbench","MM-ClawBench","agents","MiniMax","2026",62.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-multiagentbrowsecompprerelease-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-multiagentbrowsecompprerelease","Multi-Agent BrowseComp — 10-agent team prerelease configuration","agents","Anthropic","2026",93.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-multiagentbrowsecompprerelease-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-multiagentbrowsecompprerelease","Multi-Agent BrowseComp — 10-agent team prerelease configuration","agents","Anthropic","2026",93.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-officeqapro-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",57.9,63.2743,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-officeqapro-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",57.9,61.3734,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-officeqapro-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",57.9,61.3734,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-officeqapro-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",43.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-officeqapro-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",43.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-officeqapro-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",43.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-officeqapro-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",66.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-officeqapro-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",66.2,96.9957,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-officeqapro-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",66.2,96.9957,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-officeqapro-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",66.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-officeqapro-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",66.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-officeqapro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",53.2,42.4779,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-officeqapro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",53.2,41.2017,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-officeqapro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",53.2,41.2017,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-officeqapro-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",54.1,46.4602,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-officeqapro-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",54.1,45.0644,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-officeqapro-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",54.1,45.0644,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-officeqapro-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",63.3,87.1681,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-officeqapro-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",63.3,84.5494,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-officeqapro-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",63.3,84.5494,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-officeqapro-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",45.1,6.6372,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-officeqapro-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",45.1,6.4378,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-officeqapro-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors","2026",45.1,6.4378,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1955","kimi-k3","Kimi K3","Claude Code harness; OfficeQA Pro with PDF corpus rendered as images; max reasoning.",null,"Claude Code harness; OfficeQA Pro with PDF corpus rendered as images; max reasoning.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors",null,63.3,63.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 63.3 on OfficeQA Pro. Family remains not-weighted until multi-model weight review."],["evidence-2026-07-1955--configuration--kimi-k3-max","kimi-k3","Kimi K3","Claude Code harness; OfficeQA Pro with PDF corpus rendered as images; max reasoning.","kimi-k3-max","Claude Code harness; OfficeQA Pro with PDF corpus rendered as images; max reasoning.","officeqa-pro","OfficeQA Pro","agents","OfficeQA authors",null,63.3,63.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 63.3 on OfficeQA Pro. Family remains not-weighted until multi-model weight review."],["evidence-2026-07-2061","claude-opus-5","Claude Opus 5","OSWorld 2.0 computer-use evaluation as published in the Anthropic Claude Opus 5 launch table.",null,"OSWorld 2.0 computer-use evaluation as published in the Anthropic Claude Opus 5 launch table.","osworld","OSWorld","agents","OSWorld authors","2.0",70.6,70.6,"percent","higher","1.6.0","ranking-eligible","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 70.6% on OSWorld 2.0. Mapped to the active osworld family (GPT-5.6 Sol 62.6 in the same table matches existing osworld rows)."],["benchlm-ref-claude-opus-4-5-osworld-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",66.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-osworld-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",66.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-osworld-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",66.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",47.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",47.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","osworld","OSWorld","agents","OSWorld authors","2026",47.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-580","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,66.3,66.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-582","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,13.9,13.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-347","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,20.6,20.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-583","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,8.3,8.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-348","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,13,13,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-346","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,45.6,45.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-344","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,62.6,62.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-345","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,50.2,50.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-349","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,4.6,4.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-581","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,14.2,14.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-350","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld","OSWorld","agents","OSWorld authors",null,2.8,2.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-2062","claude-opus-5","Claude Opus 5","OSWorld 2.0 as published in the Anthropic Claude Opus 5 launch table.",null,"OSWorld 2.0 as published in the Anthropic Claude Opus 5 launch table.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2.0",70.6,70.6,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Directory mirror of OSWorld 2.0 under the BenchLM-named family."],["google-gemini-37-eval-osworld-2-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2.0 pre-08.08 patch",33.8,33.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-osworld-2-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2.0 pre-08.08 patch",47.9,47.9,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-osworld-2-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2.0 pre-08.08 patch",50.2,50.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-claude-opus-4-7-adaptive-osworld2-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",18.2,25.7525,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-osworld2-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",18.2,22.7139,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-osworld2-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",18.2,22.7139,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-osworld2-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13.9,18.5619,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-osworld2-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13.9,16.3717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-osworld2-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13.9,16.3717,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworld2-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",20.6,29.7659,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworld2-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",20.6,26.2537,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworld2-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",20.6,26.2537,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-osworld2-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",70.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-osworld2-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",70.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworld2-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",8.3,9.1973,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworld2-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",8.3,8.1121,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworld2-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",8.3,8.1121,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworld2-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13,17.0569,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworld2-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13,15.0442,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworld2-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",13,15.0442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-osworld2-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",45.6,71.5719,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-osworld2-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",45.6,63.1268,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-osworld2-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",45.6,63.1268,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-osworld2-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",62.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-osworld2-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",62.6,88.2006,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-osworld2-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",62.6,88.2006,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-osworld2-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",50.2,79.2642,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-osworld2-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",50.2,69.9115,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-osworld2-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",50.2,69.9115,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworld2-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,3.01,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworld2-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,2.6549,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworld2-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,2.6549,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworld2-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,3.01,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworld2-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,2.6549,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworld2-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",4.6,2.6549,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworld2-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",14.2,19.0635,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworld2-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",14.2,16.8142,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworld2-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",14.2,16.8142,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworld2-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",2.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworld2-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",2.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworld2-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","2026",2.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:osworld-binary:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Binary",2.8,2.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:osworld-binary:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Binary",19.4,19.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-osworld2:cell:vision:osworld-binary:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Binary",19.4,19.4,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["osworld2-claude-opus-4-7-max-batched-tool-500","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7; reasoning=max; toolSetting=batched tool","claude-opus-4-7-max","Claude Opus 4.7; reasoning=max; toolSetting=batched tool","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",18.2,18.2,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 48.91; dataset size 108."],["osworld2-claude-opus-4-7-max-standard-500","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7; reasoning=max; toolSetting=standard","claude-opus-4-7-max","Claude Opus 4.7; reasoning=max; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",13.9,13.9,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-16","official-leaderboard","Official binary accuracy; partial score 49.1; dataset size 108."],["osworld2-claude-opus-4-8-max-batched-tool-500","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8; reasoning=max; toolSetting=batched tool","claude-opus-4-8-max","Claude Opus 4.8; reasoning=max; toolSetting=batched tool","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",20.6,20.6,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-16","official-leaderboard","Official binary accuracy; partial score 54.8; dataset size 108."],["osworld2-claude-opus-4-8-max-standard-500","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8; reasoning=max; toolSetting=standard","claude-opus-4-8-max","Claude Opus 4.8; reasoning=max; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",18.52,18.52,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 49.33; dataset size 108."],["osworld2-claude-sonnet-4-6-max-standard-500","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6; reasoning=max; toolSetting=standard","claude-sonnet-4-6-max","Claude Sonnet 4.6; reasoning=max; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",8.3,8.3,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 41.5; dataset size 108."],["osworld2-claude-sonnet-4-6-medium-standard-500","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6; reasoning=medium; toolSetting=standard","claude-sonnet-4-6-livebench-2026-06-25-medium","Claude Sonnet 4.6; reasoning=medium; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",9.3,9.3,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 33.9; dataset size 108."],["osworld2-gpt-5-5-xhigh-batch-tool-500","gpt-5-5","GPT-5.5","GPT-5.5; reasoning=xhigh; toolSetting=batch tool","gpt-5-5-xhigh","GPT-5.5; reasoning=xhigh; toolSetting=batch tool","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",13,13,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-16","official-leaderboard","Official binary accuracy; partial score 49.5; dataset size 108."],["osworld2-kimi-k2-6-enabled-standard-500","kimi-k2-6","Kimi K2.6","Kimi 2.6; reasoning=enabled; toolSetting=standard","kimi-k2-6-osworld-enabled","Kimi 2.6; reasoning=enabled; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",4.6,4.6,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 22.1; dataset size 108."],["osworld2-minimax-m3-enabled-standard-500","minimax-m3","MiniMax M3","MiniMax M3; reasoning=enabled; toolSetting=standard","minimax-m3-osworld-enabled","MiniMax M3; reasoning=enabled; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",4.6,4.6,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 22.3; dataset size 108."],["osworld2-qwen-3-7-plus-thinking-standard-500","qwen-3-7-plus","Qwen3.7-Plus","Qwen 3.7-Plus; reasoning=thinking; toolSetting=standard","qwen-3-7-plus-epoch-qwen3-7-plus","Qwen 3.7-Plus; reasoning=thinking; toolSetting=standard","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","OSWorld 2.0; tasks v2026.06.24",2.8,2.8,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-25","2026-06-25","production::refresh-osworld-20","refresh-osworld-20","OSWorld permanent refresh source","OSWorld","https://osworld-v2.xlang.ai/","2026-06-25","2026-08-07","2026-08-17","official-leaderboard","Official binary accuracy; partial score 21.5; dataset size 108."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:osworld-partial:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Partial",21.5,21.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:osworld-partial:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Partial",48,48,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-osworld2:cell:vision:osworld-partial:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-osworld2","OSWorld 2.0","agents","Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","Partial",52.3,52.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","OSWorld 2.0 reports separate Binary and Partial-credit tracks. Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-claude-fable-osworldverified-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-osworldverified-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-osworldverified-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-osworldverified-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-osworldverified-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-osworldverified-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",85,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-osworldverified-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",66.3,59.3478,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-osworldverified-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",66.3,59.3478,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-osworldverified-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",66.3,59.3478,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-osworldverified-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.7,73.2609,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-osworldverified-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.7,73.2609,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-osworldverified-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.7,73.2609,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-osworldverified-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78,84.7826,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-osworldverified-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78,84.7826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-osworldverified-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78,84.7826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworldverified-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83.4,96.5217,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworldverified-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83.4,96.5217,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-osworldverified-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83.4,96.5217,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-osworldverified-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",61.4,48.6957,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-osworldverified-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",61.4,48.6957,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-osworldverified-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",61.4,48.6957,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworldverified-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworldverified-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-osworldverified-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-osworldverified-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",81.2,91.7391,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-osworldverified-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",81.2,91.7391,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-osworldverified-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",81.2,91.7391,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-osworldverified-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.4,85.6522,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-osworldverified-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.4,85.6522,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-osworldverified-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.4,85.6522,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",74,76.087,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",74,76.087,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",74,76.087,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-osworldverified-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83,95.6522,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-osworldverified-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83,95.6522,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-osworldverified-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",83,95.6522,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-osworldverified-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",47.3,18.0435,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-osworldverified-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",47.3,18.0435,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-osworldverified-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",47.3,18.0435,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-osworldverified-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",64.7,55.8696,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-osworldverified-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",64.7,55.8696,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-osworldverified-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",64.7,55.8696,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-osworldverified-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",75,78.2609,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-osworldverified-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",75,78.2609,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-osworldverified-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",75,78.2609,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-osworldverified-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-osworldverified-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-osworldverified-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",72.1,71.9565,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-osworldverified-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",39,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-osworldverified-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",39,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-osworldverified-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",39,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworldverified-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.7,86.3043,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworldverified-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.7,86.3043,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-osworldverified-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.7,86.3043,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-122b-a10b-osworldverified-2026-07-21","holo3-122b-a10b","Holo3-122B-A10B","Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.85,86.6304,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-122b-a10b-osworldverified-2026-07-27","holo3-122b-a10b","Holo3-122B-A10B","Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.85,86.6304,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-122b-a10b-osworldverified-2026-08-01","holo3-122b-a10b","Holo3-122B-A10B","Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",78.85,86.6304,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-35b-a3b-osworldverified-2026-07-21","holo3-35b-a3b","Holo3-35B-A3B","Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",82.56,94.6957,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-35b-a3b-osworldverified-2026-07-27","holo3-35b-a3b","Holo3-35B-A3B","Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",82.56,94.6957,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo3-35b-a3b-osworldverified-2026-08-01","holo3-35b-a3b","Holo3-35B-A3B","Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",82.56,94.6957,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworldverified-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.1,74.1304,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworldverified-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.1,74.1304,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-osworldverified-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.1,74.1304,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworldverified-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",70.06,67.5217,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworldverified-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",70.06,67.5217,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-osworldverified-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",70.06,67.5217,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworldverified-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",80.8,90.8696,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworldverified-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",80.8,90.8696,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-osworldverified-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",80.8,90.8696,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-osworldverified-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",56.2,37.3913,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-osworldverified-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",56.2,37.3913,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-osworldverified-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",56.2,37.3913,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",58,41.3043,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",58,41.3043,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",58,41.3043,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",54.5,33.6957,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",54.5,33.6957,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",54.5,33.6957,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworldverified-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.3,74.5652,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworldverified-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.3,74.5652,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-osworldverified-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","osworld-verified","OSWorld-Verified","agents","OSWorld","2025",73.3,74.5652,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-193","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",85,85,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-194","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",85,85,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-587","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",66.3,66.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-15-claude-opus-4-6-osworld-verified-verified-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",72.7,72.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-585","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",72.7,72.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-195","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",83.4,83.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-586","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",72.1,72.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-2031","claude-sonnet-5","Claude Sonnet 5","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",81.2,81.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 OSWorld-Verified comparison score."],["evidence-2026-07-196","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",81.2,81.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2050","gemini-3-flash","Gemini 3 Flash","As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.",null,"As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",65.1,65.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","OSWorld-Verified 65.1% for Gemini 3 Flash from launch comparison."],["evidence-2026-07-013","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",76.2,76.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-004","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",78.4,78.4,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-198","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",78.4,78.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2038","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. OSWorld-Verified as published in Google launch blog.",null,"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. OSWorld-Verified as published in Google launch blog.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",74,74,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","OSWorld-Verified 74.0% for 3.5 Flash-Lite."],["evidence-2026-07-1992","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). OSWorld Verified; averaged over 5 runs; max 100 steps; 1080p.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). OSWorld Verified; averaged over 5 runs; max 100 steps; 1080p.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",83,83,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","OSWorld-Verified 83.0% from Google evaluation table / launch blog."],["evidence-2026-07-202","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",64.7,64.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-199","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",75,75,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-588","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",39,39,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-023","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",78.7,78.7,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-197","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",78.7,78.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2018","gpt-5-6-luna","GPT-5.6 Luna","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",72.6,72.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna OSWorld-Verified comparison score."],["evidence-2026-07-201","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",70.06,70.06,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-584","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",80.8,80.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1903","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (OSWorld-Verified).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (OSWorld-Verified).","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",80.8,80.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for OSWorld-Verified; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1903--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (OSWorld-Verified).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (OSWorld-Verified).","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",80.8,80.8,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for OSWorld-Verified; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-08-15-qwen3-6-27b-osworld-verified-verified-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",63.9,63.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-200","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",73.3,73.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-qwen-3-7-plus-osworld-verified-verified-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",73.3,73.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-qwen-3-8-max-osworld-verified","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",86.1,86.1,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 86.1 on OSWorld-Verified."],["evidence-2026-08-qwen-3-8-max-osworld-verified--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",86.1,86.1,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 86.1 on OSWorld-Verified."],["evidence-2026-08-15-qwen-3-8-27b-osworld-verified-verified","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified",84.3,84.3,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-muse-glimmer-30b-osworld-verified-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified / 361 tasks",65.9,65.9,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported mean task reward on the 361-task split, excluding eight Google Drive tasks."],["evidence-2026-08-muse-glimmer-30b-osworld-verified-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","osworld-verified","OSWorld-Verified","agents","OSWorld","Verified / 361 tasks",65.9,65.9,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported mean task reward on the 361-task split, excluding eight Google Drive tasks."],["benchlm-ref-nemotron-3-ultra-pinchbench-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-pinchbench","PinchBench","agents","Kilo Code","2026",90,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-pinchbench-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-pinchbench","PinchBench","agents","Kilo Code","2026",90,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-pinchbench-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-pinchbench","PinchBench","agents","Kilo Code","2026",90,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["nvidia-nemotron-3-5-lightning-pinchbench-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","benchlm-pinchbench","PinchBench","agents","Kilo Code","2026",85.37,85.37,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["benchlm-ref-claude-opus-4-5-qwenclawbench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.3,4,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-qwenclawbench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.3,4,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-qwenclawbench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.3,4,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-qwenclawbench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.1,18.4,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-qwenclawbench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.1,18.4,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-qwenclawbench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.1,18.4,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-qwenclawbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.3,20,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-qwenclawbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.3,20,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-qwenclawbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",54.3,20,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",59,57.6,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",59,57.6,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",59,57.6,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-qwenclawbench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",51.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-qwenclawbench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",51.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-qwenclawbench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",51.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenclawbench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",53.4,12.8,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenclawbench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",53.4,12.8,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenclawbench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",53.4,12.8,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-qwenclawbench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",57.2,43.2,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-qwenclawbench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",57.2,43.2,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-qwenclawbench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",57.2,43.2,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.6,6.4,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.6,6.4,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",52.6,6.4,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenclawbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",64.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenclawbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",64.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenclawbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",64.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenclawbench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",61.8,80,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenclawbench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",61.8,80,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenclawbench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenclawbench","QwenClawBench","agents","Qwen","2026",61.8,80,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1532,78.9474,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1532,78.9474,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1532,78.9474,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenwebbench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1487,52.6316,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenwebbench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1487,52.6316,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-qwenwebbench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1487,52.6316,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1397,0,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1397,0,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1397,0,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenwebbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1568,100,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenwebbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1568,100,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-qwenwebbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1568,100,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenwebbench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1536,81.2865,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenwebbench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1536,81.2865,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-qwenwebbench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-qwenwebbench","QwenWebBench","agents","Qwen","2026",1536,81.2865,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-recreation-bench-2026-08-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","recreation-bench","RecreationBench","agents","Qwen","2026-08",29.8,29.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:recreation:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","recreation-bench","RecreationBench","agents","Qwen","2026-08",30.2,30.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-recreation-bench-2026-08-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","recreation-bench","RecreationBench","agents","Qwen","2026-08",30.2,30.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:recreation:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","recreation-bench","RecreationBench","agents","Qwen","2026-08",47.1,47.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-recreation-bench-2026-08","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","recreation-bench","RecreationBench","agents","Qwen","2026-08",47.1,47.1,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-opus-4-6-researchclawbench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.9,86.2069,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-researchclawbench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.9,86.2069,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-researchclawbench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.9,86.2069,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-researchclawbench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-researchclawbench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-researchclawbench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-researchclawbench-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",21.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-researchclawbench-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",21.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-researchclawbench-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",21.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-researchclawbench-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17.1,54.023,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-researchclawbench-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17.1,54.023,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-researchclawbench-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17.1,54.023,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-researchclawbench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.3,10.3448,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-researchclawbench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.3,10.3448,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-researchclawbench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.3,10.3448,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-researchclawbench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-researchclawbench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-researchclawbench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-researchclawbench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.2,66.6667,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-researchclawbench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.2,66.6667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-researchclawbench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.2,66.6667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-researchclawbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-researchclawbench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-researchclawbench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",20.7,95.4023,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-researchclawbench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-researchclawbench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-researchclawbench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-researchclawbench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17,52.8736,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-researchclawbench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17,52.8736,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-researchclawbench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",17,52.8736,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-researchclawbench-2026-07-21","grok-4-1","Grok 4.1","Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.5,12.6437,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-researchclawbench-2026-07-27","grok-4-1","Grok 4.1","Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.5,12.6437,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-researchclawbench-2026-08-01","grok-4-1","Grok 4.1","Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",13.5,12.6437,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-researchclawbench-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",12.4,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-researchclawbench-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",12.4,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-researchclawbench-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",12.4,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-researchclawbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14,18.3908,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-researchclawbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14,18.3908,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-researchclawbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14,18.3908,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-researchclawbench-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-researchclawbench-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-researchclawbench-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-researchclawbench-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-researchclawbench-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-researchclawbench-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",15.3,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-researchclawbench-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",16.9,51.7241,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-researchclawbench-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",16.9,51.7241,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-researchclawbench-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",16.9,51.7241,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-researchclawbench-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.8,85.0575,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-researchclawbench-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.8,85.0575,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-researchclawbench-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",19.8,85.0575,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-researchclawbench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14.2,20.6897,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-researchclawbench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14.2,20.6897,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-researchclawbench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",14.2,20.6897,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-researchclawbench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-researchclawbench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-researchclawbench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18,64.3678,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-researchclawbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.7,72.4138,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-researchclawbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.7,72.4138,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-researchclawbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-researchclawbench","ResearchClawBench","agents","InternScience","2026",18.7,72.4138,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-spreadsheetbench2-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-spreadsheetbench2","SpreadsheetBench 2","agents","Moonshot AI","2026",34.8,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-spreadsheetbench2-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-spreadsheetbench2","SpreadsheetBench 2","agents","Moonshot AI","2026",34.8,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-spreadsheetbench2-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-spreadsheetbench2","SpreadsheetBench 2","agents","Moonshot AI","SpreadsheetBench 2",34.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports SpreadsheetBench 2=34.8 with the Claude Code harness at max effort. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["evidence-2026-07-1983","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.0/2.1",69.4,69.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-terminal-bench-2026-07-20","benchlm-terminal-bench-2026-07-20","Terminal-Bench 2.0 Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/terminalBench","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public terminal-bench leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1985","hy3","Hy3","Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.0/2.1",54.4,54.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-terminal-bench-2026-07-20","benchlm-terminal-bench-2026-07-20","Terminal-Bench 2.0 Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/terminalBench","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public terminal-bench leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1984","qwen3-6-max","Qwen3.6 Max","Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.0/2.1",65.4,65.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-terminal-bench-2026-07-20","benchlm-terminal-bench-2026-07-20","Terminal-Bench 2.0 Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/terminalBench","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public terminal-bench leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-058","claude-fable-5","Claude Fable 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.64,84.64,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-fable-5-evals","aa-claude-fable-5-evals","Artificial Analysis evaluations for claude-fable-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-033","claude-fable-5","Claude Fable 5","Claude Code; reasoning xhigh; Terminal-Bench 2.1 verified submission.",null,"Claude Code; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.82,83.82,"percent","higher","1.0.0","ranking-eligible","direct","2026-06-07","2026-06-07","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-06-07","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 83.8 to 83.82 on 2026-08-16."],["evidence-2026-07-033--configuration--claude-fable-5-xhigh","claude-fable-5","Claude Fable 5","Claude Code; reasoning xhigh; Terminal-Bench 2.1 verified submission.","claude-fable-5-xhigh","Claude Code; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.8,83.8,"percent","higher","1.0.0","reference-only","direct","2026-06-07","2026-06-07","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-06-07","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["aa-individual:claude-fable-5:terminal-bench:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.6441947565543,84.6441947565543,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-claude-fable-5-terminal-bench-2-1-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88,88,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-07-152","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.3,84.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88,88,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-07-034","claude-fable-5","Claude Fable 5","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.45,80.45,"percent","higher","1.0.0","ranking-eligible","direct","2026-06-05","2026-06-05","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-06-05","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 80.4 to 80.45 on 2026-08-16."],["evidence-2026-07-034--configuration--claude-fable-5-high","claude-fable-5","Claude Fable 5","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","claude-fable-5-high","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.4,80.4,"percent","higher","1.0.0","reference-only","direct","2026-06-05","2026-06-05","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-06-05","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-149","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88,88,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-claude-opus-4-6-terminal-bench-2-1-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.2,78.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen3.8-27B text table, Terminal Bench 2.1 (Terminus). Muse Glimmer 51.7 is already stored from the Muse owner card and is not duplicated."],["aa-current:claude-opus-4-7-adaptive:terminal-bench:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.14606741573,83.14606741573,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["terminal-bench-21-claude-opus-4-7-claude-code-max","claude-opus-4-7","Claude Opus 4.7","Opus 4.7; reasoning=max; agent=Claude Code","claude-opus-4-7-max","Opus 4.7; reasoning=max; agent=Claude Code","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",68.9,68.9,"percent","higher","2.2.0","ranking-eligible","direct","2026-05-01","2026-07-08","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-08","2026-07-16","2026-08-16","official-leaderboard","New exact owner row absent from source-native production data; owner record fdb8393b-5b29-4645-b784-84f52cf31722; https://github.com/harbor-framework/terminal-bench-2-1/pull/44."],["terminal-bench-21-claude-opus-4-7-terminus-2-max","claude-opus-4-7","Claude Opus 4.7","Opus 4.7; reasoning=max; agent=Terminus 2","claude-opus-4-7-max","Opus 4.7; reasoning=max; agent=Terminus 2","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",66.07,66.07,"percent","higher","2.2.0","ranking-eligible","direct","2026-05-01","2026-07-08","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-08","2026-07-16","2026-08-16","official-leaderboard","New exact owner row absent from source-native production data; owner record f867b631-36c8-476e-aa2f-96007ae70da0; https://github.com/harbor-framework/terminal-bench-2-1/pull/46."],["evidence-2026-07-061","claude-opus-4-8","Claude Opus 4.8","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.64,84.64,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-opus-4-8-evals","aa-claude-opus-4-8-evals","Artificial Analysis evaluations for claude-opus-4-8","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-038","claude-opus-4-8","Claude Opus 4.8","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.88,78.88,"percent","higher","1.0.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 78.9 to 78.88 on 2026-08-16."],["evidence-2026-07-038--configuration--claude-opus-4-8-high","claude-opus-4-8","Claude Opus 4.8","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","claude-opus-4-8-high","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.9,78.9,"percent","higher","1.0.0","reference-only","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",85,85,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["aa-individual:claude-opus-4-8:terminal-bench:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.6441947565543,84.6441947565543,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-claude-opus-4-8-terminal-bench-2-1-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",85,85,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-07-157","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",74.6,74.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["aa-individual:claude-opus-5:terminal-bench:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",89.1385767790262,89.1385767790262,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-claude-opus-5","claude-opus-5","Claude Opus 5","Claude Opus 5 (reasoning configuration not stated in chart)","claude-opus-5-meta-muse-12-release-unspecified","Claude Opus 5 (reasoning configuration not stated in chart)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",86.7,86.7,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-1723","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",36.3296,36.3296,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1377","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",55.8052,55.8052,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-065","claude-sonnet-5","Claude Sonnet 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.52,80.52,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-sonnet-5-evals","aa-claude-sonnet-5-evals","Artificial Analysis evaluations for claude-sonnet-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-041","claude-sonnet-5","Claude Sonnet 5","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",74.61,74.61,"percent","higher","1.0.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 74.6 to 74.61 on 2026-08-16."],["evidence-2026-07-041--configuration--claude-sonnet-5-high","claude-sonnet-5","Claude Sonnet 5","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","claude-sonnet-5-high","Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",74.6,74.6,"percent","higher","1.0.0","reference-only","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["google-gemini-37-eval-terminal-bench-2-1-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.4,80.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["aa-individual:claude-sonnet-5:terminal-bench:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.5243445692884,80.5243445692884,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-155","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.4,80.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2029","claude-sonnet-5","Claude Sonnet 5","Terminus-2; as published in Gemini 3.6 Flash evaluation table.",null,"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.4,80.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 Terminal-Bench comparison score."],["evidence-2026-07-1631","command-a-plus","Command A+","Command A+",null,"Command A+","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",22.8464,22.8464,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1537","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",44.9438,44.9438,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1522","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",46.8165,46.8165,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-083","deepseek-v4-flash","DeepSeek V4 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",61.8,61.8,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-flash-evals","aa-deepseek-v4-flash-evals","Artificial Analysis evaluations for deepseek-v4-flash","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",61.8,61.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-07-162","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",49.1,49.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["deepseek-v4-flash-0731-terminal-bench-2-1-2026-07-31","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-0731-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.7,82.7,"percent","higher","2.3.0","reference-only","direct","2026-07-31","2026-07-31","production::deepseek-v4-flash-0731-model-card","deepseek-v4-flash-0731-model-card","DeepSeek-V4-Flash-0731 official model card and provider evaluation","DeepSeek","https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","2026-07-31","2026-08-24","2026-08-24","provider-reported","Source-native DeepSeek subject value with the official code-agent configuration; the exact Terminal-Bench agent scaffold and track remain unresolved."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.7,82.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-flash-vision-exp-terminal-bench-2-1-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.9,83.9,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value. Terminal-Bench 2.1 identity and metric are resolved, but the source does not resolve the required category-local agent scaffold and track, so this remains visible non-scoring evidence."],["aa-individual:deepseek-v4-flash-vision-exp:terminal-bench:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",74.1573033707865,74.1573033707865,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual terminal-bench result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["evidence-2026-07-080","deepseek-v4-pro","DeepSeek V4 Pro","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",64.04,64.04,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-pro-evals","aa-deepseek-v4-pro-evals","Artificial Analysis evaluations for deepseek-v4-pro","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",72.1,72.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-07-161","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",59.1,59.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",87.9,87.9,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["aa-individual:deepseek-v4-pro-0813:terminal-bench:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.65168539325839,78.65168539325839,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-deepseek-v4-pro-0813-terminal-bench-2-1-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",87.9,87.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-07-1797","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",28.4644,28.4644,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1927","gemini-3-pro","Gemini 3 Pro Preview","Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.8,65.8,"percent","higher","1.4.1","excluded","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-16","excluded","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-07-1927--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","gemini-3-pro-high","Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.8,65.8,"percent","higher","1.4.1","excluded","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-16","excluded","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-07-1926","gemini-3-pro","Gemini 3 Pro Preview","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",73.9,73.9,"percent","higher","1.4.1","excluded","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-16","excluded","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-07-1926--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","gemini-3-pro-high","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",73.9,73.9,"percent","higher","1.4.1","excluded","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-16","excluded","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-07-2051","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","As published in Gemini 3.5 Flash-Lite launch materials (prior Flash-Lite comparison).",null,"As published in Gemini 3.5 Flash-Lite launch materials (prior Flash-Lite comparison).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",31,31,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Terminal-Bench 2.1 31% for 3.1 Flash-Lite from launch comparison."],["evidence-2026-07-073","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",73.78,73.78,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gemini-3-1-pro-preview-evals","aa-gemini-3-1-pro-preview-evals","Artificial Analysis evaluations for gemini-3-1-pro-preview","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-1-pro-preview","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-042","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.8,65.8,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-043","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.6,65.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-031","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.",null,"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",70.3,70.3,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-2013","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Terminus-2; as published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro.",null,"Terminus-2; as published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",73.8,73.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Terminal-Bench 2.1 comparison row for 3.1 Pro from 3.6 table."],["evidence-2026-07-156","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.2,76.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-030","gemini-3-5-flash","Gemini 3.5 Flash","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.",null,"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.2,76.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-030--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","gemini-3-5-flash-high","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.2,76.2,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-2034","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. Terminal-Bench 2.1 as published in Google launch blog.",null,"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. Terminal-Bench 2.1 as published in Google launch blog.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",54,54,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Terminal-Bench 2.1 54% for 3.5 Flash-Lite."],["evidence-2026-07-1998","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis Terminal-Bench v2.1 independent evaluation (as listed on AA / BenchLM).",null,"Artificial Analysis Terminal-Bench v2.1 independent evaluation (as listed on AA / BenchLM).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.5,77.5,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA independent Terminal-Bench v2.1 score 77.5%."],["google-gemini-37-eval-terminal-bench-2-1-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78,78,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["aa-individual:gemini-3-6-flash:terminal-bench:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.52808988764049,77.52808988764049,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-gemini-3-6-flash","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (reasoning configuration not stated in chart)","gemini-3-6-flash-meta-muse-12-release-unspecified","Gemini 3.6 Flash (reasoning configuration not stated in chart)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.9,78.9,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-1990","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Terminus-2 harness; Terminal-Bench 2.1.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Terminus-2 harness; Terminal-Bench 2.1.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78,78,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Terminal-Bench 2.1 Terminus-2 score from Google evaluation table."],["google-gemini-37-eval-terminal-bench-2-1-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",85.8,85.8,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["aa-individual:gemini-3-7-flash:terminal-bench:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.2771535580524,78.2771535580524,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:terminal-bench:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",4.494382022472,4.494382022472,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1617","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",38.9513,38.9513,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["aa-current:gemma-4-26b-a4b:terminal-bench:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",38.951310861423,38.951310861423,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1738","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",49.4382,49.4382,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1551","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",45.3184,45.3184,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1928","glm-5-1","GLM-5.1","Claude Code; reasoning max; Terminal-Bench 2.1 verified submission.",null,"Claude Code; reasoning max; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",58.7,58.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-16","official-leaderboard","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-07-589","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81,81,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81,81,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["aa-current:glm-5-2:terminal-bench:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.902621722846,77.902621722846,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["evidence-2026-08-15-glm-5-2-terminal-bench-2-1-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81,81,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["aa-individual:glm-5-3:terminal-bench:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.8951310861423,83.8951310861423,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-glm-5-3-terminal-bench-2-1","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.2,88.2,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["aa-individual:glm-5-3-flash:terminal-bench:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.2696629213483,84.2696629213483,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual terminal-bench result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:terminal-bench:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",10.112359550562,10.112359550562,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:terminal-bench:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",3.74531835206,3.74531835206,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1337","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",35.206,35.206,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1337--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",35.206,35.206,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-103","gpt-5-4","GPT-5.4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.28,78.28,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-4-evals","aa-gpt-5-4-evals","Artificial Analysis evaluations for gpt-5-4","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["aa-individual:gpt-5-4:terminal-bench:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.2771535580524,78.2771535580524,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1287","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",59.176,59.176,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1287--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",59.176,59.176,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-114","gpt-5-5","GPT-5.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.27,84.27,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-5-evals","aa-gpt-5-5-evals","Artificial Analysis evaluations for gpt-5-5","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-035","gpt-5-5","GPT-5.5","Codex; reasoning xhigh; Terminal-Bench 2.1 verified submission.",null,"Codex; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.15,83.15,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 83.1 to 83.15 on 2026-08-16."],["evidence-2026-07-035--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","Codex; reasoning xhigh; Terminal-Bench 2.1 verified submission.","gpt-5-5-xhigh","Codex; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.1,83.1,"percent","higher","1.0.0","reference-only","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-154","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82,82,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["aa-individual:gpt-5-5:terminal-bench:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.2696629213483,84.2696629213483,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-036","gpt-5-5","GPT-5.5","Terminus 2; reasoning xhigh; Terminal-Bench 2.1 verified submission.",null,"Terminus 2; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.98,77.98,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 78 to 77.98 on 2026-08-16."],["evidence-2026-07-036--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","Terminus 2; reasoning xhigh; Terminal-Bench 2.1 verified submission.","gpt-5-5-xhigh","Terminus 2; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78,78,"percent","higher","1.0.0","reference-only","direct","2026-05-01","2026-05-01","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-05-01","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-032","gpt-5-5","GPT-5.5","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.",null,"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.2,78.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-032--configuration--gpt-5-5-high","gpt-5-5","GPT-5.5","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","gpt-5-5-high","Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.2,78.2,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-055","gpt-5-6-luna","GPT-5.6 Luna","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.9,80.9,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-luna-evals","aa-gpt-5-6-luna-evals","Artificial Analysis evaluations for gpt-5-6-luna","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-040","gpt-5-6-luna","GPT-5.6 Luna","Codex; reasoning max; Terminal-Bench 2.1 verified submission.",null,"Codex; reasoning max; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",75.73,75.73,"percent","higher","1.0.0","ranking-eligible","direct","2026-07-11","2026-07-11","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-11","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 75.7 to 75.73 on 2026-08-16."],["evidence-2026-07-040--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","Codex; reasoning max; Terminal-Bench 2.1 verified submission.","gpt-5-6-luna-max","Codex; reasoning max; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",75.7,75.7,"percent","higher","1.0.0","reference-only","direct","2026-07-11","2026-07-11","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-11","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-151","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.7,84.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["aa-individual:gpt-5-6-luna:terminal-bench:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.8988764044944,80.8988764044944,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-2016","gpt-5-6-luna","GPT-5.6 Luna","Terminus-2; as published in Gemini 3.6 Flash evaluation table.",null,"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",84.7,84.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna Terminal-Bench comparison score."],["evidence-2026-07-047","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.01,88.01,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-sol-evals","aa-gpt-5-6-sol-evals","Artificial Analysis evaluations for gpt-5-6-sol","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-148","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",91.9,91.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["aa-individual:gpt-5-6-sol:terminal-bench:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.0149812734082,88.0149812734082,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-gpt-5-6-sol-terminal-bench-2-1-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.8,88.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-07-051","gpt-5-6-terra","GPT-5.6 Terra","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.01,88.01,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-terra-evals","aa-gpt-5-6-terra-evals","Artificial Analysis evaluations for gpt-5-6-terra","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-039","gpt-5-6-terra","GPT-5.6 Terra","Codex; reasoning max; Terminal-Bench 2.1 verified submission.",null,"Codex; reasoning max; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.43,78.43,"percent","higher","1.0.0","ranking-eligible","direct","2026-07-11","2026-07-11","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-11","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 78.4 to 78.43 on 2026-08-16."],["evidence-2026-07-039--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","Codex; reasoning max; Terminal-Bench 2.1 verified submission.","gpt-5-6-terra-max","Codex; reasoning max; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",78.4,78.4,"percent","higher","1.0.0","reference-only","direct","2026-07-11","2026-07-11","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-11","2026-07-16","2026-07-15","official-leaderboard","Verified by the Terminal-Bench maintainers."],["evidence-2026-07-150","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",87.4,87.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["google-gemini-37-eval-terminal-bench-2-1-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",87.4,87.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["aa-individual:gpt-5-6-terra:terminal-bench:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.0149812734082,88.0149812734082,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-gpt-5-6-terra","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (reasoning configuration not stated in chart)","gpt-5-6-terra-meta-muse-12-release-unspecified","GPT-5.6 Terra (reasoning configuration not stated in chart)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81.8,81.8,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-077","grok-4-3","Grok 4.3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",39.7,39.7,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis evaluations for grok-4-3","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-121","grok-4-5","Grok 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81.65,81.65,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-5-evals","aa-grok-4-5-evals","Artificial Analysis evaluations for grok-4-5","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-037","grok-4-5","Grok 4.5","Cursor CLI; reasoning high; Terminal-Bench 2.1 verified submission.",null,"Cursor CLI; reasoning high; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",79.33,79.33,"percent","higher","1.0.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers. Owner precision refreshed from 79.3 to 79.33 on 2026-08-16."],["evidence-2026-07-153","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.3,83.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["aa-individual:grok-4-5:terminal-bench:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81.6479400749064,81.6479400749064,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-grok-4-5","grok-4-5","Grok 4.5","Grok 4.5 (reasoning configuration not stated in chart)","grok-4-5-meta-muse-12-release-unspecified","Grok 4.5 (reasoning configuration not stated in chart)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",81.6,81.6,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-2023","grok-4-5","Grok 4.5","Terminus-2; as published in Gemini 3.6 Flash evaluation table.",null,"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",83.3,83.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Grok 4.5 Terminal-Bench comparison score."],["aa-individual:grok-4-6:terminal-bench:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.38951310861421,88.38951310861421,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:kimi-k2-5-reasoning:terminal-bench:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",45.692883895131,45.692883895131,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1415","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.9176,65.9176,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1433","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",67.4157,67.4157,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["deepseek-v4-pro-0813-release-terminal-bench-2-1-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.3,88.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["aa-individual:kimi-k3:terminal-bench:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",85.0187265917603,85.0187265917603,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-kimi-k3-terminal-bench-2-1-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.3,88.3,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-07-1936","kimi-k3","Kimi K3","KimiCode harness; reasoning max; Terminal-Bench 2.1 as published in the Kimi K3 launch blog.",null,"KimiCode harness; reasoning max; Terminal-Bench 2.1 as published in the Kimi K3 launch blog.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.3,88.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 88.3 on Terminal-Bench 2.1 with KimiCode at max reasoning effort."],["evidence-2026-07-1936--configuration--kimi-k3-max","kimi-k3","Kimi K3","KimiCode harness; reasoning max; Terminal-Bench 2.1 as published in the Kimi K3 launch blog.","kimi-k3-max","KimiCode harness; reasoning max; Terminal-Bench 2.1 as published in the Kimi K3 launch blog.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",88.3,88.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 88.3 on Terminal-Bench 2.1 with KimiCode at max reasoning effort."],["aa-current:llama-4-maverick:terminal-bench:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",7.865168539326,7.865168539326,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:terminal-bench:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",3.74531835206,3.74531835206,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:terminal-bench:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",50.187265917603,50.187265917603,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["evidence-2026-08-15-longcat-2-0-terminal-bench-2-1","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",70.8,70.8,"percent","higher","2.1.0","ranking-eligible","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score. Starred competitor cells were not admitted."],["evidence-2026-07-1574","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",63.6704,63.6704,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-591","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",68.4,68.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-110","minimax-m3","MiniMax M3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",65.17,65.17,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-minimax-m3-evals","aa-minimax-m3-evals","Artificial Analysis evaluations for minimax-m3","Artificial Analysis","https://artificialanalysis.ai/models/minimax-m3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-160","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",66,66,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-096","mistral-large-3","Mistral Large 3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",11.99,11.99,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-large-3-evals","aa-mistral-large-3-evals","Artificial Analysis evaluations for mistral-large-3","Artificial Analysis","https://artificialanalysis.ai/models/mistral-large-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-087","mistral-medium-3-5","Mistral Medium 3.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",50.56,50.56,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-medium-3-5-evals","aa-mistral-medium-3-5-evals","Artificial Analysis evaluations for mistral-medium-3-5","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["aa-current:mistral-medium-3-5-128b:terminal-bench:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",50.561797752809,50.561797752809,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-091","mistral-small-4","Mistral Small 4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",20.97,20.97,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-small-4-evals","aa-mistral-small-4-evals","Artificial Analysis evaluations for mistral-small-4","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["aa-current:mistral-small-4-reasoning:terminal-bench:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",20.973782771536,20.973782771536,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-08-muse-glimmer-30b-terminal-bench-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",51.7,51.7,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies the Terminus 2 result as sourced from Artificial Analysis."],["evidence-2026-08-muse-glimmer-30b-terminal-bench-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",51.7,51.7,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies the Terminus 2 result as sourced from Artificial Analysis."],["evidence-2026-07-590","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80,80,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1925","muse-spark-1-1","Muse Spark 1.1","mini-SWE-agent; reasoning xhigh; Terminal-Bench 2.1 verified submission.",null,"mini-SWE-agent; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.18,76.18,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-08-16","official-leaderboard","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard. Owner precision refreshed from 76.2 to 76.18 on 2026-08-16."],["evidence-2026-07-1925--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","mini-SWE-agent; reasoning xhigh; Terminal-Bench 2.1 verified submission.","muse-spark-1-1-xhigh","mini-SWE-agent; reasoning xhigh; Terminal-Bench 2.1 verified submission.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.2,76.2,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::terminal-bench-21","terminal-bench-21","Terminal-Bench 2.1 leaderboard","Terminal-Bench","https://www.tbench.ai/leaderboard/terminal-bench/2.1","2026-07-09","2026-07-16","2026-07-16","official-leaderboard","Verified by the Terminal-Bench maintainers on the public 2.1 leaderboard."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-muse-spark-1-1","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (reasoning configuration not stated in chart)","muse-spark-1-1-meta-muse-12-release-unspecified","Muse Spark 1.1 (reasoning configuration not stated in chart)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",76.2,76.2,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["aa-individual:muse-spark-1-1:terminal-bench:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.9026217228464,77.9026217228464,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1915","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.9,77.9,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1915--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",77.9,77.9,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1902","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Terminal-Bench 2.0).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Terminal-Bench 2.0).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80,80,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Terminal-Bench 2.0; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1902--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Terminal-Bench 2.0).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Terminal-Bench 2.0).","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80,80,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Terminal-Bench 2.0; retained with provider-reported provenance via Meta evaluation report."],["google-gemini-37-eval-terminal-bench-2-1-muse-spark-1-2-2026-08-13","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2","muse-spark-1-2-xhigh","Muse Spark 1.2","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.9,82.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["aa-individual:muse-spark-1-2:terminal-bench:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",80.1498127340824,80.1498127340824,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual terminal-bench result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["evidence-2026-08-15-meta-muse-12-terminal-bench-2-1-muse-spark-1-2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.9,82.9,"percent","higher","2.0.0","reference-only","direct","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-terminal-bench-chart","refresh-meta-muse-spark-1-2-terminal-bench-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/terminal-bench-2-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Exact Muse Spark 1.2 result retained with the published xhigh setting."],["evidence-2026-08-muse-spark-1-2-terminal-bench-2-1","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) with Muse Code",null,"Muse Spark 1.2 (xhigh) with Muse Code","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.9,82.9,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 82.9% using Muse Code, averaged over five attempts across all 89 Terminal-Bench 2.1 tasks."],["evidence-2026-08-muse-spark-1-2-terminal-bench-2-1--configuration--muse-spark-1-2-xhigh","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) with Muse Code","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh) with Muse Code","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",82.9,82.9,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 82.9% using Muse Code, averaged over five attempts across all 89 Terminal-Bench 2.1 tasks."],["aa-current:nemotron-3-nano-30b:terminal-bench:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",6.741573033708,6.741573033708,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1824","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",38.5768,38.5768,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1602","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",53.9326,53.9326,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1644","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",29.588,29.588,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-terminal-bench-2-1-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",24.58,24.58,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1498","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",47.5655,47.5655,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1480","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",51.3109,51.3109,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["aa-current:qwen3-5-397b-reasoning:terminal-bench:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",51.310861423221,51.310861423221,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:terminal-bench:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",47.565543071161,47.565543071161,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1458","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",60.6742,60.6742,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-08-15-qwen3-6-27b-terminal-bench-2-1-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",63.4,63.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen3.8-27B text table, Terminal Bench 2.1 (Terminus). Muse Glimmer 51.7 is already stored from the Muse owner card and is not duplicated."],["aa-current:qwen3-6-35b-a3b:terminal-bench:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",44.943820224719,44.943820224719,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-099","qwen-3-7-max","Qwen3.7-Max","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",74.53,74.53,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-qwen3-7-max-evals","aa-qwen3-7-max-evals","Artificial Analysis evaluations for qwen3-7-max","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-7-max","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-159","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",69.7,69.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-158","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",70.3,70.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-qwen-3-7-plus-terminal-bench-2-1-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",64,64,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen3.8-27B text table, Terminal Bench 2.1 (Terminus). Muse Glimmer 51.7 is already stored from the Muse owner card and is not duplicated."],["evidence-2026-08-qwen-3-8-max-terminal-bench-2-1","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",86.6,86.6,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 86.6. Evaluated with Claude Code, average over 10 runs, five-hour timeout, and max_tokens 131,072."],["evidence-2026-08-qwen-3-8-max-terminal-bench-2-1--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",86.6,86.6,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 86.6. Evaluated with Claude Code, average over 10 runs, five-hour timeout, and max_tokens 131,072."],["evidence-2026-08-15-qwen-3-8-max-terminal-bench-2-1-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",86.6,86.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207, temp=1.0, top_p=1, max_new_tokens=65536, 6h timeout."],["evidence-2026-08-15-qwen-3-8-27b-terminal-bench-2-1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",73,73,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen3.8-27B text table, Terminal Bench 2.1 (Terminus). Muse Glimmer 51.7 is already stored from the Muse owner card and is not duplicated."],["aa-current:qwen-3-8-flash-next:terminal-bench:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",86.142322097378,86.142322097378,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:terminal-bench:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2.1",20.59925093633,20.59925093633,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["benchlm-ref-claude-fable-terminalbench2-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",84.3,86.4769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-terminalbench2-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",84.3,86.4769,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-terminalbench2-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",88,93.0605,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-terminalbench2-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",88,93.0605,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-terminalbench2-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.3,41.9929,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-terminalbench2-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.3,41.9929,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-terminalbench2-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.4,52.847,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-terminalbench2-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.4,52.847,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-terminalbench2-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.4,59.9644,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-terminalbench2-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.4,59.9644,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbench2-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",74.6,69.2171,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbench2-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",74.6,69.2171,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50,25.4448,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50,25.4448,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.1,41.637,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.1,41.637,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-terminalbench2-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80.4,79.5374,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-terminalbench2-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80.4,79.5374,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-terminalbench2-2026-07-21","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",61.7,46.2633,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-terminalbench2-2026-07-27","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",61.7,46.2633,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-terminalbench2-2026-07-21","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.3,59.7865,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-terminalbench2-2026-07-27","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.3,59.7865,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.6,37.1886,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.6,37.1886,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.6,37.1886,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.6,37.1886,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.9,37.7224,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.9,37.7224,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-terminalbench2-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",49.1,23.8434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-terminalbench2-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",49.1,23.8434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.9,37.7224,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.9,37.7224,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.3,49.1103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.3,49.1103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.3,49.1103,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.3,49.1103,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",67.9,57.2954,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",67.9,57.2954,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-terminalbench2-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.1,41.637,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-terminalbench2-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.1,41.637,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",67.9,57.2954,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",67.9,57.2954,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-terminalbench2-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",76.2,72.0641,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-terminalbench2-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",76.2,72.0641,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",54,32.5623,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",54,32.5623,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-terminalbench2-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",41,9.4306,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-terminalbench2-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",41,9.4306,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-terminalbench2-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.2,36.4769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-terminalbench2-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.2,36.4769,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-terminalbench2-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.5,49.4662,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-terminalbench2-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.5,49.4662,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbench2-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",81,80.605,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbench2-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",81,80.605,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-terminalbench2-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",77.3,74.0214,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-terminalbench2-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",77.3,74.0214,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-terminalbench2-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",75.1,70.1068,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-terminalbench2-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",75.1,70.1068,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-terminalbench2-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",60,43.2384,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-terminalbench2-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",60,43.2384,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-terminalbench2-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",46.3,18.8612,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-terminalbench2-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",46.3,18.8612,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-terminalbench2-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",82,82.3843,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-terminalbench2-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",82,82.3843,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-terminalbench2-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",84.7,87.1886,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-terminalbench2-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",84.7,87.1886,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbench2-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",91.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbench2-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",91.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbench2-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",87.4,91.9929,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbench2-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",87.4,91.9929,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-terminalbench2-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",47.1,20.2847,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-terminalbench2-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",47.1,20.2847,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-terminalbench2-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",83.3,84.6975,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-terminalbench2-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",83.3,84.6975,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-terminalbench2-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",54.4,33.274,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-terminalbench2-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",54.4,33.274,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-terminalbench2-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.8,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-terminalbench2-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",63.8,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-terminalbench2-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50.8,26.8683,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-terminalbench2-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50.8,26.8683,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-terminalbench2-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50.8,26.8683,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-terminalbench2-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",50.8,26.8683,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-terminalbench2-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",66.7,55.1601,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-terminalbench2-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",66.7,55.1601,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-terminalbench2-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",88.3,93.5943,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-terminalbench2-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",88.3,93.5943,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-terminalbench2-2026-07-21","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",45.8,17.9715,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-terminalbench2-2026-07-27","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",45.8,17.9715,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-terminalbench2-2026-07-21","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",70.2,61.3879,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-terminalbench2-2026-07-27","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",70.2,61.3879,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-terminalbench2-2026-07-21","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",35.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-terminalbench2-2026-07-27","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",35.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-terminalbench2-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",46,18.3274,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-terminalbench2-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",46,18.3274,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-terminalbench2-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.8,53.5587,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-terminalbench2-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.8,53.5587,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",68.4,58.1851,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",68.4,58.1851,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-terminalbench2-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",57,37.9004,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-terminalbench2-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",57,37.9004,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbench2-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",66,53.9146,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbench2-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",66,53.9146,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-terminalbench2-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59,41.4591,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-terminalbench2-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59,41.4591,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-terminalbench2-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80,78.8256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-terminalbench2-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80,78.8256,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbench2-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.4,36.8327,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbench2-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",56.4,36.8327,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-terminalbench2-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",64.2,50.7117,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-terminalbench2-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",64.2,50.7117,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-terminalbench2-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",77.5,74.3772,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-terminalbench2-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",77.5,74.3772,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-terminalbench2-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",43.1,13.1673,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-terminalbench2-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",43.1,13.1673,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.4,52.847,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",65.4,52.847,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-terminalbench2-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",41.6,10.4982,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-terminalbench2-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",41.6,10.4982,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-terminalbench2-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",52.5,29.8932,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-terminalbench2-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",52.5,29.8932,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",49.4,24.3772,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",49.4,24.3772,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",40.5,8.5409,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",40.5,8.5409,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-terminalbench2-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.3,41.9929,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-terminalbench2-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.3,41.9929,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-terminalbench2-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",61.6,46.0854,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-terminalbench2-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",61.6,46.0854,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",51.5,28.1139,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",51.5,28.1139,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbench2-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.7,60.4982,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbench2-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",69.7,60.4982,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-terminalbench2-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",70.3,61.5658,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-terminalbench2-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",70.3,61.5658,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-terminalbench2-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80.2,79.1815,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-terminalbench2-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",80.2,79.1815,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",82.1,82.5623,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",82.1,82.5623,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-terminalbench2-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.5,42.3488,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-terminalbench2-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",59.5,42.3488,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-terminalbench2-2026-07-21","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",81.5,81.4947,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-terminalbench2-2026-07-27","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench","Terminal-Bench","agents","Terminal-Bench team","2026",81.5,81.4947,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-fable-5-terminal-bench-3-3-0-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",33.7,33.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["xai-grok-4-6-release-terminal-bench-v3-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",34.1,34.1,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-08-15-claude-opus-4-8-terminal-bench-3-3-0-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",21.1,21.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["google-gemini-37-eval-terminal-bench-3-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",14.6,14.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-terminal-bench-3-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",5.4,5.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-terminal-bench-3-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-high","Gemini 3.7 Flash","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",14.9,14.9,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["evidence-2026-08-15-glm-5-2-terminal-bench-3-3-0-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",4.6,4.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["evidence-2026-08-15-glm-5-3-terminal-bench-3-3-0","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",28.3,28.3,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["evidence-2026-08-15-gpt-5-6-sol-terminal-bench-3-3-0-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",34.6,34.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["xai-grok-4-6-release-terminal-bench-v3-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",34.6,34.6,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["google-gemini-37-eval-terminal-bench-3-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",20.8,20.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["xai-grok-4-6-release-terminal-bench-v3-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",15.7,15.7,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-terminal-bench-v3-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",26,26,"percent","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["evidence-2026-08-15-kimi-k3-terminal-bench-3-3-0-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","terminal-bench-3","Terminal-Bench 3.0","agents","Harbor / Terminal-Bench","3.0",17.4,17.4,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: Claude Code 2.1.207 avg@3, 400K context, 128K max output, 600 turns, 10h timeout. Scaffold is left unspecified so the reviewed Terminal-Bench 3 default track is used."],["benchlm-ref-claude-fable-terminalbenchhard-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",62.9,92.665,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-terminalbenchhard-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",62.9,92.665,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-terminalbenchhard-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",62.9,92.9245,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",58.3,81.4181,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",58.3,81.4181,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",58.3,82.0755,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-terminalbenchhard-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",25,0,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-terminalbenchhard-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",25,0,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-terminalbenchhard-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",25,3.5377,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,25.9169,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,25.9169,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,28.5377,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,25.9169,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,25.9169,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",35.6,28.5377,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,51.8337,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,51.8337,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,53.5377,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,51.8337,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,51.8337,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",46.2,53.5377,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-terminalbenchhard-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",53.8,70.4156,"percent","higher","1.5.0","excluded","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-terminalbenchhard-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",53.8,70.4156,"percent","higher","1.6.0","excluded","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-terminalbenchhard-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",40.9,38.8753,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-terminalbenchhard-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,27.8729,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-terminalbenchhard-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,27.8729,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-terminalbenchhard-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,30.4245,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbenchhard-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,63.0807,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbenchhard-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,63.0807,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbenchhard-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,64.3868,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-terminalbenchhard-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",60.6,87.0416,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-terminalbenchhard-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",60.6,87.0416,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",65.9,100,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",65.9,100,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",65.9,100,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",57.6,79.7066,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",57.6,79.7066,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",57.6,80.4245,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-terminalbenchhard-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",23.5,0,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-terminalbenchhard-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",43.9,46.2103,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",43.2,44.4988,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",43.2,44.4988,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",43.2,46.4623,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbenchhard-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",42.4,42.5428,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbenchhard-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",42.4,42.5428,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbenchhard-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",42.4,44.5755,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",33.3,20.2934,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",33.3,20.2934,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",33.3,23.1132,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,27.8729,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,27.8729,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",36.4,30.4245,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbenchhard-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,63.0807,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbenchhard-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,63.0807,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbenchhard-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","2026",50.8,64.3868,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-command-a-plus-terminal-bench-hard-hard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",25,25,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-terminal-bench-hard-hard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",34.1,34.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-terminal-bench-hard-hard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",41.7,41.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-terminal-bench-hard-hard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",33.3,33.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-terminal-bench-hard-hard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",2.3,2.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-terminal-bench-hard-hard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis","Hard",28.3,28.3,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-910","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,62.8788,62.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-910--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,62.8788,62.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1193","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,27.2727,27.2727,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1401","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.0606,31.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1389","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,34.3352,34.3352,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-989","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (Reasoning)",null,"Claude Opus 4.5 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.9697,46.9697,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1036","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.2121,46.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1036--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.2121,46.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1024","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,51.5152,51.5152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1024--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,51.5152,51.5152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1176","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,58.3333,58.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1176--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,58.3333,58.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1773","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,21.2121,21.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1722","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.0606,31.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1376","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,35.6061,35.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1014","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,53.0303,53.0303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1014--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,53.0303,53.0303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1630","command-a-plus","Command A+","Command A+",null,"Command A+","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,25,25,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1536","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,30.303,30.303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1521","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,35.6061,35.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1160","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,35.6061,35.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1160--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,35.6061,35.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1103","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.2121,46.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1103--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.2121,46.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1835","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,16.6667,16.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1796","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,26.5152,26.5152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1134","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,38.6364,38.6364,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1211","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)",null,"Gemini 3 Pro Preview (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,41.6667,41.6667,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","excluded","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1211--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)","gemini-3-pro-high","Gemini 3 Pro Preview (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,41.6667,41.6667,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","excluded","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1075","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,24.2424,24.2424,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1218","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,53.7879,53.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-919","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,40.9091,40.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-919--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,40.9091,40.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1879","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,18.1818,18.1818,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1616","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,13.64,13.64,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1229","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,36.3636,36.3636,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1737","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,25,25,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1550","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.8182,31.8182,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1030","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.1818,43.1818,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-967","glm-5-turbo","GLM-5-Turbo","GLM-5-Turbo",null,"GLM-5-Turbo","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,33.3333,33.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1090","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.1818,43.1818,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1255","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,50.7576,50.7576,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1255--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,50.7576,50.7576,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-984","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,32.5758,32.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1336","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,32.5758,32.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1336--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,32.5758,32.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1322","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.8788,37.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1322--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.8788,37.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1349","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,28.7879,28.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1349--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,28.7879,28.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1068","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,45.4545,45.4545,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1068--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,45.4545,45.4545,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1301","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.9697,46.9697,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1301--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.9697,46.9697,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1312","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.1212,37.1212,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1312--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.1212,37.1212,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1083","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)",null,"GPT-5.3 Codex (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,53.0303,53.0303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1083--configuration--gpt-5-3-codex-xhigh","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)","gpt-5-3-codex-xhigh","GPT-5.3 Codex (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,53.0303,53.0303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1051","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,57.5758,57.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1051--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,57.5758,57.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1286","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,52.2727,52.2727,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1286--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,52.2727,52.2727,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1149","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,42.4242,42.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1149--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,42.4242,42.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-974","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,60.6061,60.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-974--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,60.6061,60.6061,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-947","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,65.9091,65.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-947--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,65.9091,65.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-996","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,57.5758,57.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-996--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,57.5758,57.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1868","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,17.4242,17.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1656","grok-4","Grok 4","Grok 4",null,"Grok 4","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.8788,37.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1760","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,18.9394,18.9394,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1445","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.8788,37.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-1114","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.8788,37.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1890","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,17.4242,17.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1847","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,23.4848,23.4848,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1668","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.0606,31.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-1185","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,34.8485,34.8485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1414","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.9394,43.9394,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1432","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,44.697,44.697,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1954","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis Terminal-Bench Hard / Terminal-Bench v2.1 independent evaluation.",null,"Kimi K3; Artificial Analysis Terminal-Bench Hard / Terminal-Bench v2.1 independent evaluation.","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,85,85,"percent","higher","1.4.1","reference-only","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis terminal evaluation score 85% as mirrored on BenchLM for Kimi K3."],["evidence-2026-07-1785","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.0606,31.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1585","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,28.0303,28.0303,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1169","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,34.8485,34.8485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1096","mimo-v2-pro","MiMo-V2-Pro","MiMo-V2-Pro",null,"MiMo-V2-Pro","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,40.9091,40.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1573","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,41.6667,41.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-930","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.1818,43.1818,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1749","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,25.7576,25.7576,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1688","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,28.7879,28.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1562","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,34.8485,34.8485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-1059","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,39.3939,39.3939,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1005","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,42.4242,42.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1044","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,15.9091,15.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1245","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,33.3333,33.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-955","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,17.4242,17.4242,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1266","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,45.4545,45.4545,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1823","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,28.7879,28.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1601","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,36.3636,36.3636,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1643","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,24.2424,24.2424,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1858","o1","o1","o1",null,"o1","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,12.8788,12.8788,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1362","o3","o3","o3",null,"o3","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,37.1212,37.1212,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1809","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,15.1515,15.1515,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1809--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,15.1515,15.1515,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1679","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,24.2424,24.2424,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1497","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,31.0606,31.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1508","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,32.5758,32.5758,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1709","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,26.5152,26.5152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1479","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,40.9091,40.9091,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1699","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,21.2121,21.2121,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1457","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,34.8485,34.8485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1468","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.9394,43.9394,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-1273","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,43.9394,43.9394,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1125","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,50.7576,50.7576,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["evidence-2026-07-1203","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","terminal-bench-hard","Terminal-Bench Hard","agents","Laude Institute / Artificial Analysis",null,46.9697,46.9697,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-terminalbench-hard","aa-terminalbench-hard","Terminal-Bench Hard Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/terminalbench-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Terminal-Bench Hard accuracy from Artificial Analysis."],["benchlm-ref-claude-opus-4-5-toolathlon-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-toolathlon-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-toolathlon-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-toolathlon-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",59.9,67.7618,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-toolathlon-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",59.9,67.7618,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-toolathlon-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",59.9,67.7618,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-toolathlon-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",43.5,34.0862,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-toolathlon-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",40.7,28.3368,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-toolathlon-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",40.7,28.3368,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-toolathlon-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",40.7,28.3368,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",47.8,42.9158,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-toolathlon-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49,45.3799,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-toolathlon-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-toolathlon-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-toolathlon-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",51.8,51.1294,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-toolathlon-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",56.5,60.7803,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-toolathlon-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",56.5,60.7803,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-toolathlon-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",56.5,60.7803,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-toolathlon-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",38,22.7926,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-toolathlon-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",38,22.7926,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-toolathlon-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",38,22.7926,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-toolathlon-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",48.2,43.7372,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-toolathlon-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",48.2,43.7372,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-toolathlon-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",48.2,43.7372,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-toolathlon-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",54.6,56.8789,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-toolathlon-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",54.6,56.8789,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-toolathlon-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",54.6,56.8789,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-toolathlon-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",42.9,32.8542,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-toolathlon-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",42.9,32.8542,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-toolathlon-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",42.9,32.8542,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-toolathlon-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",35.5,17.6591,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-toolathlon-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",35.5,17.6591,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-toolathlon-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",35.5,17.6591,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-toolathlon-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",55.6,58.9322,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-toolathlon-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",55.6,58.9322,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-toolathlon-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",55.6,58.9322,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-toolathlon-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.4,54.4148,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-toolathlon-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.4,54.4148,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-toolathlon-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.4,54.4148,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-toolathlon-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",58,63.8604,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-toolathlon-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",58,63.8604,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-toolathlon-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",58,63.8604,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-toolathlon-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.1,53.7988,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-toolathlon-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.1,53.7988,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-toolathlon-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",53.1,53.7988,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-toolathlon-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",27.8,1.848,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-toolathlon-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",27.8,1.848,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-toolathlon-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",27.8,1.848,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-toolathlon-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",50,47.4333,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-toolathlon-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",50,47.4333,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-toolathlon-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",50,47.4333,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-toolathlon-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-toolathlon-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-toolathlon-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",46.3,39.8357,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-toolathlon-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",75.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-toolathlon-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",75.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-toolathlon-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",75.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-toolathlon-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",36.3,19.3018,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-toolathlon-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",36.3,19.3018,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-toolathlon-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",36.3,19.3018,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-toolathlon-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",39.8,26.4887,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-toolathlon-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",39.8,26.4887,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-toolathlon-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",39.8,26.4887,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",26.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",26.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",26.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-toolathlon-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49.5,46.4066,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-toolathlon-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49.5,46.4066,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-toolathlon-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","toolathlon","Toolathlon","agents","Google","2026",49.5,46.4066,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-657","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",43.5,43.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-250","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",59.9,59.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-258","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",40.7,40.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["deepseek-v4-flash-vision-exp-toolathlon-verified-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","toolathlon","Toolathlon","agents","Google","May 2026",75.9,75.9,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value for the canonical Toolathlon-Verified May 2026 protocol."],["evidence-2026-07-257","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",46.3,46.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-003","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","toolathlon","Toolathlon","agents","Google","May 2026",56.5,56.5,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-252","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",56.5,56.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-659","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",38,38,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-655","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",48.2,48.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-254","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",54.6,54.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-660","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",35.5,35.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-022","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","toolathlon","Toolathlon","agents","Google","May 2026",55.6,55.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-253","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",55.6,55.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-255","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",53.4,53.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-251","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",58,58,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-256","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",53.1,53.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-259","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",27.8,27.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1939","kimi-k3","Kimi K3","Toolathlon-Verified; max reasoning; as published in the Kimi K3 launch blog.",null,"Toolathlon-Verified; max reasoning; as published in the Kimi K3 launch blog.","toolathlon","Toolathlon","agents","Google","May 2026",73.2,73.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 73.2 on Toolathlon-Verified."],["evidence-2026-07-1939--configuration--kimi-k3-max","kimi-k3","Kimi K3","Toolathlon-Verified; max reasoning; as published in the Kimi K3 launch blog.","kimi-k3-max","Toolathlon-Verified; max reasoning; as published in the Kimi K3 launch blog.","toolathlon","Toolathlon","agents","Google","May 2026",73.2,73.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 73.2 on Toolathlon-Verified."],["evidence-2026-07-656","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",46.3,46.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-654","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",75.6,75.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1905","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Toolathlon).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Toolathlon).","toolathlon","Toolathlon","agents","Google","May 2026",75.6,75.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Toolathlon; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1905--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Toolathlon).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Toolathlon).","toolathlon","Toolathlon","agents","Google","May 2026",75.6,75.6,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Toolathlon; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-658","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","toolathlon","Toolathlon","agents","Google","May 2026",39.8,39.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-15-claude-fable-5-toolathlon-verified-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",74.7,74.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-claude-opus-4-8-toolathlon-verified-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",76.2,76.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-deepseek-v4-pro-0813-toolathlon-verified-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",74.1,74.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-glm-5-2-toolathlon-verified-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",59.9,59.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-glm-5-3-toolathlon-verified","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","toolathlon","Toolathlon","agents","Google","Verified",73,73,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-gpt-5-6-sol-toolathlon-verified-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",74.9,74.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-kimi-k3-toolathlon-verified-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",76.5,76.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["evidence-2026-08-15-qwen-3-8-max-toolathlon-verified-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","toolathlon","Toolathlon","agents","Google","Verified",72.5,72.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official evaluation service, pass@1 averaged over 3 runs."],["benchlm-ref-claude-opus-5-toolathlonverifiedavgturns-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedavgturns","Toolathlon Verified average assistant turns","agents","Anthropic","2026",23.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverifiedavgturns-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedavgturns","Toolathlon Verified average assistant turns","agents","Anthropic","2026",23.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverifiedpass3all-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedpass3all","Toolathlon Verified Pass cubed","agents","Anthropic","2026",73.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverifiedpass3all-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedpass3all","Toolathlon Verified Pass cubed","agents","Anthropic","2026",73.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverifiedpass3-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedpass3","Toolathlon Verified Pass@3","agents","Anthropic","2026",87,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverifiedpass3-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverifiedpass3","Toolathlon Verified Pass@3","agents","Anthropic","2026",87,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverified-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",80.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-toolathlonverified-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",80.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlonverified-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",70.3,66.6667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-toolathlonverified-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",70.3,66.6667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-toolathlonverified-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",54.4,15.2104,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-toolathlonverified-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",73.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-toolathlonverified-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",73.2,76.0518,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-toolathlonverified-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",73.2,76.0518,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-toolathlonverified-2026-07-21","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",49.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-toolathlonverified-2026-07-27","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",49.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-toolathlonverified-2026-08-01","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","2026",49.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-toolathlon-verified-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",77.9,77.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",76.2,76.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",49.7,49.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-flash-0731-toolathlon-verified-2026-07-31","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-0731-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",70.3,70.3,"percent","higher","2.3.0","reference-only","direct","2026-07-31","2026-07-31","production::deepseek-v4-flash-0731-model-card","deepseek-v4-flash-0731-model-card","DeepSeek-V4-Flash-0731 official model card and provider evaluation","DeepSeek","https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","2026-07-31","2026-08-24","2026-08-24","provider-reported","Source-native DeepSeek subject value for Toolathlon-Verified with the official code-agent configuration."],["deepseek-v4-pro-0813-release-toolathlon-verified-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",70.3,70.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",55.9,55.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",74.1,74.1,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",59.9,59.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-toolathlon-verified-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","August 2026 verified",76.5,76.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:toolathlon:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","Pass@1",70.3,70.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:toolathlon:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","Pass@1",50.6,50.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:toolathlon:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","Pass@1",67.1,67.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-toolathlonverified:cell:language:toolathlon:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-toolathlonverified","Toolathlon-Verified","agents","Moonshot AI","Pass@1",73.5,73.5,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","pass@1 Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-agents-a1-vitabench-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",38.75,71.7593,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-vitabench-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",38.75,71.7593,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-vitabench-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",38.75,71.7593,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vitabench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",23.3,24.0741,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vitabench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",23.3,24.0741,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vitabench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",23.3,24.0741,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-vitabench-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",17,4.6296,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-vitabench-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",17,4.6296,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-vitabench-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",17,4.6296,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-vitabench-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",18.5,9.2593,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-vitabench-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",18.5,9.2593,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-vitabench-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",18.5,9.2593,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-vitabench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",15.5,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-vitabench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",15.5,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-vitabench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",15.5,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vitabench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",43.7,87.037,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vitabench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",43.7,87.037,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vitabench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",43.7,87.037,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vitabench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",44.3,88.8889,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vitabench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",44.3,88.8889,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vitabench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",44.3,88.8889,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",35.6,62.037,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",35.6,62.037,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",35.6,62.037,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-vitabench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",47.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-vitabench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",47.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-vitabench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",47.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-vitabench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",45.6,92.9012,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-vitabench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",45.6,92.9012,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-vitabench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vitabench","VITA-Bench","agents","Meituan LongCat Team","2025",45.6,92.9012,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-561","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","webarena","WebArena","agents","WebArena authors",null,69,69,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1906","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (WebArena-Verified).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (WebArena-Verified).","webarena","WebArena","agents","WebArena authors",null,69,69,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for WebArena-Verified; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1906--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (WebArena-Verified).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (WebArena-Verified).","webarena","WebArena","agents","WebArena authors",null,69,69,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for WebArena-Verified; retained with provider-reported provenance via Meta evaluation report."],["benchlm-ref-muse-spark-1-1-webarenaverified-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-webarenaverified","WebArena-Verified Browser Agent Benchmark","agents","Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","2025",69,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-webarenaverified-verified-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-webarenaverified","WebArena-Verified Browser Agent Benchmark","agents","Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","Verified",48.8,48.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-webarenaverified-verified-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-webarenaverified","WebArena-Verified Browser Agent Benchmark","agents","Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","Verified",55.3,55.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-webarenaverified-verified","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-webarenaverified","WebArena-Verified Browser Agent Benchmark","agents","Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","Verified",64.8,64.8,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-opus-4-5-wideresearch-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",76.4,78.744,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-wideresearch-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",76.4,78.744,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-wideresearch-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",76.4,78.744,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-wideresearch-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",69.8,46.8599,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-wideresearch-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",69.8,46.8599,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-wideresearch-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",69.8,46.8599,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-wideresearch-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",72.7,60.8696,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-wideresearch-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",72.7,60.8696,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-wideresearch-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",72.7,60.8696,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-wideresearch-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",80.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-wideresearch-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",80.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-wideresearch-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",80.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-wideresearch-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74,67.1498,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-wideresearch-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74,67.1498,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-wideresearch-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74,67.1498,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-wideresearch-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74.3,68.599,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-wideresearch-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74.3,68.599,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-wideresearch-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",74.3,68.599,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",60.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",60.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-wideresearch","WideResearch","agents","Qwen","2026",60.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-276","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,98.5,98.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-706","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,86.3,86.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-707","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,84.8,84.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-712","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,74,74,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-282","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,94.4,94.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-710","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,79.5,79.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-714","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,43.3,43.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-705","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,87.1,87.1,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-291","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,31.3,31.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-279","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,95.6,95.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-280","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,95.3,95.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-713","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,59.9,59.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-698","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,98.2,98.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-697","glm-5-turbo","GLM-5-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,98.5,98.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-699","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,97.7,97.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-695","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,99.1,99.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-696","glm-5v-turbo","GLM-5V-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,98.5,98.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-709","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,81.9,81.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-288","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,86,86,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-286","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,87.1,87.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-711","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,76,76,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-283","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,93.9,93.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-289","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,85.1,85.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-287","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,86.3,86.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-277","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,97.7,97.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-278","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,95.9,95.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-704","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,91.2,91.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-701","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,95,95,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-702","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,94.2,94.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-708","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,84.8,84.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-285","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,88.9,88.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-292","mistral-large-3","Mistral Large 3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,24.6,24.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-290","mistral-small-4","Mistral Small 4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,41.2,41.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-703","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,91.5,91.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-700","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,97.7,97.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-281","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,94.7,94.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-284","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","tau-bench","τ-bench","agents","Sierra Research",null,93,93,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["benchlm-ref-zaya1-74b-preview-tau2airline-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-tau2airline","τ²-Bench Airline Domain","agents","Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","2025",56.1,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-tau2airline-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-tau2airline","τ²-Bench Airline Domain","agents","Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","2025",56.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-tau2airline-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-tau2airline","τ²-Bench Airline Domain","agents","Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","2025",56.1,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-908","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",98.538,98.538,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-908--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",98.538,98.538,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1191","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",54.6784,54.6784,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1400","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",73.3918,73.3918,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1388","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",71.3661,71.3661,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-988","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (Reasoning)",null,"Claude Opus 4.5 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",89.4737,89.4737,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1035","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.1053,92.1053,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1035--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.1053,92.1053,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1022","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",88.5965,88.5965,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1022--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",88.5965,88.5965,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1174","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.4444,94.4444,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1174--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.4444,94.4444,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1772","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",54.6784,54.6784,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1720","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",64.6199,64.6199,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1374","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",78.0702,78.0702,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1012","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",75.731,75.731,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1012--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",75.731,75.731,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1628","command-a-plus","Command A+","Command A+",null,"Command A+","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",80.7018,80.7018,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1534","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",37.1345,37.1345,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1519","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",90.6433,90.6433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1158","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.0292,95.0292,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1158--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.0292,95.0292,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1101","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",96.1988,96.1988,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1101--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",96.1988,96.1988,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1834","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",45.614,45.614,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1794","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",54.0936,54.0936,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1132","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",80.4094,80.4094,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1210","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)",null,"Gemini 3 Pro Preview (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",87.1345,87.1345,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","excluded","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1210--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)","gemini-3-pro-high","Gemini 3 Pro Preview (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",87.1345,87.1345,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","excluded","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1073","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",31.2865,31.2865,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1216","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.614,95.614,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-917","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.3216,95.3216,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-917--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.3216,95.3216,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1878","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",36.2573,36.2573,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1614","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",43.5673,43.5673,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1227","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",59.9415,59.9415,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1735","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",70.4678,70.4678,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1548","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.9064,95.9064,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1029","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",98.2456,98.2456,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-966","glm-5-turbo","GLM-5-Turbo","GLM-5-Turbo",null,"GLM-5-Turbo","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",98.538,98.538,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1088","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",97.6608,97.6608,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1253","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",99.1228,99.1228,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1253--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",99.1228,99.1228,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-983","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",98.538,98.538,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1334","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",84.7953,84.7953,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1334--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",84.7953,84.7953,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1321","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",86.8421,86.8421,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1321--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",86.8421,86.8421,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1348","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",71.0526,71.0526,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1348--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",71.0526,71.0526,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1066","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",81.8713,81.8713,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1066--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",81.8713,81.8713,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1300","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",84.7953,84.7953,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1300--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",84.7953,84.7953,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1311","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.1053,92.1053,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1311--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.1053,92.1053,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1082","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)",null,"GPT-5.3 Codex (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",85.9649,85.9649,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1082--configuration--gpt-5-3-codex-xhigh","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)","gpt-5-3-codex-xhigh","GPT-5.3 Codex (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",85.9649,85.9649,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1049","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",87.1345,87.1345,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1049--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",87.1345,87.1345,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1284","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",83.3333,83.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1284--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",83.3333,83.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1147","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",76.0234,76.0234,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1147--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",76.0234,76.0234,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-972","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",93.8596,93.8596,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-972--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",93.8596,93.8596,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-945","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",85.0877,85.0877,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-945--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",85.0877,85.0877,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-994","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",86.2573,86.2573,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-994--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",86.2573,86.2573,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1867","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",90.3509,90.3509,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1655","grok-4","Grok 4","Grok 4",null,"Grok 4","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",74.8538,74.8538,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1759","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",65.7895,65.7895,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1444","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.9825,92.9825,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-1112","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",97.6608,97.6608,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1889","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",75.731,75.731,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1846","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",73.3918,73.3918,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1667","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.9825,92.9825,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-1183","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.9064,95.9064,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1412","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.9064,95.9064,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1430","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",90.0585,90.0585,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1784","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",89.7661,89.7661,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1584","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.0292,95.0292,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1168","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",91.2281,91.2281,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1095","mimo-v2-pro","MiMo-V2-Pro","MiMo-V2-Pro",null,"MiMo-V2-Pro","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.0292,95.0292,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1571","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",90.6433,90.6433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-928","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.152,94.152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1748","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",86.8421,86.8421,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1687","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",85.3801,85.3801,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1561","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.3216,95.3216,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-1057","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",84.7953,84.7953,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1003","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",88.8889,88.8889,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1042","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",24.5614,24.5614,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1243","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.152,94.152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-953","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",41.2281,41.2281,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1264","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",91.5205,91.5205,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1821","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",67.8363,67.8363,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1599","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",83.3333,83.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1641","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.6901,92.6901,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1857","o1","o1","o1",null,"o1","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",62.5731,62.5731,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1361","o3","o3","o3",null,"o3","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",80.7018,80.7018,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1808","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",55.5556,55.5556,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1808--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",55.5556,55.5556,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1678","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",83.6257,83.6257,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1495","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",93.5673,93.5673,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1507","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",93.8596,93.8596,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1708","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",89.1813,89.1813,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1477","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.614,95.614,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1698","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",88.3041,88.3041,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1455","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.152,94.152,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1467","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",95.9064,95.9064,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-1271","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",97.6608,97.6608,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1123","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",94.7368,94.7368,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["evidence-2026-07-1201","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2",92.9825,92.9825,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau2-bench","aa-tau2-bench","τ²-Bench Telecom Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau2-bench","2026-07-15","2026-07-15","2026-07-15","source-checked","τ²-Bench Telecom success rate from Artificial Analysis."],["benchlm-ref-claude-3-haiku-tau2bench-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",21.1,21.2916,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-tau2bench-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",21.1,21.2916,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-tau2bench-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",21.1,21.2916,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-tau2bench-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.3,52.775,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-tau2bench-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.3,52.775,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-tau2bench-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.3,52.775,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",71.4,72.0484,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",71.4,72.0484,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",71.4,72.0484,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-tau2bench-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-tau2bench-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-tau2bench-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-tau2bench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-tau2bench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-tau2bench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.5,90.3128,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.5,90.3128,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.5,90.3128,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-tau2bench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-tau2bench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-tau2bench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-tau2bench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-tau2bench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-tau2bench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-tau2bench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.6,89.4046,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-tau2bench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.6,89.4046,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-tau2bench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.6,89.4046,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-tau2bench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74,74.672,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-tau2bench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74,74.672,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-tau2bench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74,74.672,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-tau2bench-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.4,95.2573,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-tau2bench-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.4,95.2573,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-tau2bench-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.4,95.2573,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-tau2bench-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",79.5,80.222,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-tau2bench-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",79.5,80.222,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-tau2bench-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",79.5,80.222,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-tau2bench-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",80.7,81.4329,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-tau2bench-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",80.7,81.4329,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-tau2bench-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",85,85.7719,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-tau2bench-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-tau2bench-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-tau2bench-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-tau2bench-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",37.4,37.7397,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-tau2bench-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",37.4,37.7397,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-tau2bench-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",37.4,37.7397,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-tau2bench-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.8,35.116,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-tau2bench-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.8,35.116,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-tau2bench-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.8,35.116,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-tau2bench-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",78.9,79.6165,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-tau2bench-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",78.9,79.6165,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-tau2bench-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",78.9,79.6165,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-tau2bench-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-tau2bench-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",96.2,97.0737,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-tau2bench-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.5,36.8315,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-tau2bench-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.5,36.8315,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-tau2bench-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.5,36.8315,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.5,20.6862,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.5,20.6862,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.5,20.6862,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-tau2bench-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",4.1,4.1372,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-tau2bench-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",4.1,4.1372,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-tau2bench-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",4.1,4.1372,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-tau2bench-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.9,15.0353,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-tau2bench-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.9,15.0353,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-tau2bench-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.9,15.0353,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-tau2bench-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",54.1,54.5913,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-tau2bench-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",54.1,54.5913,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-tau2bench-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",54.1,54.5913,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-tau2bench-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.3,43.6932,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-tau2bench-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.3,43.6932,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-tau2bench-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.3,43.6932,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-tau2bench-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",87.1,87.891,"percent","higher","1.5.0","excluded","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-tau2bench-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",87.1,87.891,"percent","higher","1.6.0","excluded","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-tau2bench-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",87.1,87.891,"percent","higher","1.8.0","excluded","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.3,31.5843,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.3,31.5843,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.3,31.5843,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-tau2bench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.5.0","excluded","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-tau2bench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.6.0","excluded","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-tau2bench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.8.0","excluded","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-tau2bench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-tau2bench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-tau2bench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-tau2bench-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",10.5,10.5954,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-tau2bench-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",10.5,10.5954,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-tau2bench-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",10.5,10.5954,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-tau2bench-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.3,36.6297,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-tau2bench-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.3,36.6297,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-tau2bench-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",36.3,36.6297,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.6,43.996,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.6,43.996,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",43.6,43.996,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-tau2bench-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",59.9,60.444,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-tau2bench-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",59.9,60.444,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-tau2bench-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",59.9,60.444,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-tau2bench-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-tau2bench-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-tau2bench-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-tau2bench-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-tau2bench-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-tau2bench-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",20.8,20.9889,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-tau2bench-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.5,46.9223,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-tau2bench-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.5,46.9223,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-tau2bench-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.5,46.9223,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-tau2bench-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",76.9,77.5984,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-tau2bench-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",76.9,77.5984,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-tau2bench-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",76.9,77.5984,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-tau2bench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-tau2bench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-tau2bench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau2bench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.2,99.0918,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau2bench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.2,99.0918,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau2bench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.2,99.0918,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-tau2bench-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-tau2bench-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-tau2bench-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau2bench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau2bench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau2bench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-tau2bench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",99.1,100,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-tau2bench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",99.1,100,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-tau2bench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",99.1,100,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-tau2bench-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-tau2bench-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-tau2bench-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-tau2bench-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",47.1,47.5277,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-tau2bench-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",47.1,47.5277,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-tau2bench-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",47.1,47.5277,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-tau2bench-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.9,53.3804,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-tau2bench-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.9,53.3804,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-tau2bench-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",52.9,53.3804,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-tau2bench-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.3,17.4571,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-tau2bench-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.3,17.4571,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-tau2bench-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.3,17.4571,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-tau2bench-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",25.1,25.328,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-tau2bench-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",25.1,25.328,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-tau2bench-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",25.1,25.328,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-tau2bench-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-tau2bench-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.5,87.2856,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-tau2bench-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",81.9,82.6438,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-tau2bench-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",81.9,82.6438,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-tau2bench-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",81.9,82.6438,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-tau2bench-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-tau2bench-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-tau2bench-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83,83.7538,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-tau2bench-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-tau2bench-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-tau2bench-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-tau2bench-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-tau2bench-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-tau2bench-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.1,92.9364,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-tau2bench-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-tau2bench-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-tau2bench-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-tau2bench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",87.1,87.891,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-tau2bench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",87.1,87.891,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-tau2bench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.9,99.7982,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-tau2bench-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.3,84.0565,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-tau2bench-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.3,84.0565,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-tau2bench-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.4,94.2482,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-tau2bench-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",76,76.6902,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-tau2bench-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",76,76.6902,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-tau2bench-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",92.5,93.3401,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-tau2bench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.9,94.7528,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-tau2bench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.9,94.7528,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-tau2bench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98,98.89,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-tau2bench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",85.1,85.8729,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-tau2bench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",85.1,85.8729,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-tau2bench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",85.1,85.8729,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-tau2bench-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-tau2bench-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-tau2bench-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86.3,87.0838,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-tau2bench-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-tau2bench-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-tau2bench-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-tau2bench-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",60.2,60.7467,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-tau2bench-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",60.2,60.7467,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-tau2bench-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",60.2,60.7467,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-tau2bench-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-tau2bench-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-tau2bench-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",22.8,23.0071,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-tau2bench-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",13.2,13.3199,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-tau2bench-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",13.2,13.3199,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-tau2bench-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",13.2,13.3199,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-tau2bench-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19.6,19.778,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-tau2bench-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19.6,19.778,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-tau2bench-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19.6,19.778,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-tau2bench-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.6,14.7326,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-tau2bench-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.6,14.7326,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-tau2bench-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14.6,14.7326,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-tau2bench-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.9,75.5802,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-tau2bench-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.9,75.5802,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-tau2bench-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.9,75.5802,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",65.8,66.3976,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-tau2bench-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",63.7,64.2785,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-tau2bench-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",63.7,64.2785,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-tau2bench-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",63.7,64.2785,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.3,94.1473,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.3,94.1473,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.3,94.1473,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-tau2bench-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-tau2bench-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-tau2bench-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-tau2bench-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",75.7,76.3875,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-tau2bench-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",75.7,76.3875,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-tau2bench-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",75.7,76.3875,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-tau2bench-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-tau2bench-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-tau2bench-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-tau2bench-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",61.1,61.6549,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-tau2bench-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",61.1,61.6549,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-tau2bench-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",61.1,61.6549,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-tau2bench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-tau2bench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-tau2bench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau2bench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau2bench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau2bench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-tau2bench-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-tau2bench-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-tau2bench-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-tau2bench-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-tau2bench-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-tau2bench-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",16.1,16.2462,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",16.1,16.2462,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.07,88.8698,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",8.5,8.5772,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",8.5,8.5772,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",8.5,8.5772,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-tau2bench-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-tau2bench-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-tau2bench-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",86,86.781,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-tau2bench-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19,19.1726,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-tau2bench-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19,19.1726,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-tau2bench-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",19,19.1726,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-tau2bench-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.8,17.9617,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-tau2bench-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.8,17.9617,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-tau2bench-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",17.8,17.9617,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-tau2bench-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",15.5,15.6408,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-tau2bench-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",15.5,15.6408,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-tau2bench-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",15.5,15.6408,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-tau2bench-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.9,84.662,"percent","higher","1.5.0","excluded","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-tau2bench-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.9,84.662,"percent","higher","1.6.0","excluded","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-tau2bench-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.9,84.662,"percent","higher","1.8.0","excluded","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-tau2bench-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.2,92.0283,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-tau2bench-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.2,92.0283,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-tau2bench-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.2,92.0283,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-tau2bench-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-tau2bench-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-tau2bench-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95,95.8628,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau2bench-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau2bench-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau2bench-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-tau2bench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-tau2bench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-tau2bench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",84.8,85.5701,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-tau2bench-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.9,89.7074,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-tau2bench-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.9,89.7074,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-tau2bench-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",88.9,89.7074,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-tau2bench-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",30.7,30.9788,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-tau2bench-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",30.7,30.9788,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-tau2bench-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",30.7,30.9788,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-tau2bench-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.6,24.8234,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-tau2bench-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.6,24.8234,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-tau2bench-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.6,24.8234,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-tau2bench-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.3,24.5207,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-tau2bench-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.3,24.5207,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-tau2bench-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",24.3,24.5207,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-tau2bench-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-tau2bench-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-tau2bench-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-tau2bench-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-tau2bench-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-tau2bench-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",41.2,41.5742,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-tau2bench-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.5,92.331,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-tau2bench-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.5,92.331,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-tau2bench-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",91.5,92.331,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",40.9,41.2714,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",40.9,41.2714,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",40.9,41.2714,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",45.3,45.7114,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",45.3,45.7114,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",45.3,45.7114,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau2bench-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.3,84.0565,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau2bench-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.3,84.0565,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau2bench-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",83.3,84.0565,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-tau2bench-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",11.4,11.5035,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-tau2bench-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",11.4,11.5035,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-tau2bench-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",11.4,11.5035,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-tau2bench-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14,14.1271,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-tau2bench-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14,14.1271,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-tau2bench-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",14,14.1271,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-tau2bench-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",62.6,63.1685,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-tau2bench-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",62.6,63.1685,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-tau2bench-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",62.6,63.1685,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-tau2bench-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",80.7,81.4329,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-tau2bench-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",80.7,81.4329,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-tau2bench-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",80.7,81.4329,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-tau2bench-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",28.7,28.9606,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-tau2bench-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",28.7,28.9606,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-tau2bench-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",28.7,28.9606,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-tau2bench-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",0,0,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-tau2bench-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",0,0,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-tau2bench-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",0,0,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-tau2bench-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-tau2bench-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-tau2bench-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.9,96.7709,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-tau2bench-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-tau2bench-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-tau2bench-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",74.3,74.9748,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-tau2bench-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.9,94.7528,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-tau2bench-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.9,94.7528,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-tau2bench-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.9,94.7528,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-tau2bench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-tau2bench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-tau2bench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau2bench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau2bench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau2bench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.6,96.4682,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.6,94.4501,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.6,94.4501,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93.6,94.4501,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.2,90.0101,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.2,90.0101,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",89.2,90.0101,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-tau2bench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-tau2bench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-tau2bench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.2,95.0555,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau2bench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau2bench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau2bench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",97.7,98.5873,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",95.3,96.1655,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-tau2bench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.7,95.56,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-tau2bench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.7,95.56,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-tau2bench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",94.7,95.56,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-tau2bench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93,93.8446,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-tau2bench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93,93.8446,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-tau2bench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",93,93.8446,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-tau2bench-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.8,47.225,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-tau2bench-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.8,47.225,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-tau2bench-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",46.8,47.225,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-tau2bench-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.5,34.8133,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-tau2bench-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.5,34.8133,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-tau2bench-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",34.5,34.8133,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-tau2bench-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.9,32.1897,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-tau2bench-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.9,32.1897,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-tau2bench-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",31.9,32.1897,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-tau2bench-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-tau2bench-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-tau2bench-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",98.5,99.3946,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-tau2bench-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-tau2bench-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-tau2bench-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-tau2bench-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.5.0","reference-only","direct","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-tau2bench-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.6.0","reference-only","direct","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-tau2bench-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","tau2-bench","τ²-Bench Telecom","agents","Sierra / Artificial Analysis","2025",90.1,90.9183,"percent","higher","1.8.0","reference-only","direct","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-909","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",26.8041,26.8041,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-909--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",26.8041,26.8041,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:claude-fable-5:tau3-banking:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",38.144329896907195,38.144329896907195,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1192","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",9.0722,9.0722,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1023","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",28.866,28.866,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1023--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",28.866,28.866,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:claude-opus-4-7-adaptive:tau3-banking:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",34.639175257732,34.639175257732,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["evidence-2026-07-1175","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",27.6289,27.6289,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1175--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",27.6289,27.6289,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:claude-opus-4-8:tau3-banking:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",34.226804123711304,34.226804123711304,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:tau3-banking:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",42.0618556701031,42.0618556701031,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1721","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.8144,13.8144,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1375","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",18.9691,18.9691,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1013","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",30.5155,30.5155,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1013--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",30.5155,30.5155,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-960","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",28.2474,28.2474,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-960--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",28.2474,28.2474,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:claude-sonnet-5:tau3-banking:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",37.319587628866,37.319587628866,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1629","command-a-plus","Command A+","Command A+",null,"Command A+","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.7732,5.7732,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1535","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",15.8763,15.8763,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1520","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",18.7629,18.7629,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1159","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",22.8866,22.8866,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1159--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",22.8866,22.8866,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:deepseek-v4-flash-vision-exp:tau3-banking:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",41.0309278350515,41.0309278350515,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual tau3-banking result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["evidence-2026-07-1102","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.7732,25.7732,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1102--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.7732,25.7732,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:deepseek-v4-pro-0813:tau3-banking:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",39.587628865979404,39.587628865979404,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["evidence-2026-07-1795","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",9.2784,9.2784,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1133","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",17.5258,17.5258,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1074","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",8.6598,8.6598,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1217","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",16.4948,16.4948,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-918","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.3608,25.3608,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-918--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.3608,25.3608,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-2045","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis τ³-Banking independent evaluation.",null,"Artificial Analysis τ³-Banking independent evaluation.","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",16.5,16.5,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","τ³-Banking 16.5%."],["evidence-2026-07-2005","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis τ³-Banking independent evaluation.",null,"Artificial Analysis τ³-Banking independent evaluation.","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",24.5,24.5,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","τ³-Banking 24.5%."],["aa-individual:gemini-3-6-flash:tau3-banking:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",29.8969072164948,29.8969072164948,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:tau3-banking:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",35.4639175257732,35.4639175257732,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:tau3-banking:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",0.824742268041,0.824742268041,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1615","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",11.7526,11.7526,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["aa-current:gemma-4-26b-a4b:tau3-banking:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",11.958762886598,11.958762886598,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1228","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",15.0515,15.0515,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1736","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",10.5155,10.5155,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1549","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",9.6907,9.6907,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1089","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",11.5464,11.5464,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1254","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",26.8041,26.8041,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1254--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",26.8041,26.8041,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:glm-5-2:tau3-banking:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",34.639175257732,34.639175257732,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:tau3-banking:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",50.3092783505155,50.3092783505155,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:tau3-banking:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",47.2164948453608,47.2164948453608,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual tau3-banking result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:tau3-banking:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.360824742268,5.360824742268,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:tau3-banking:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",3.505154639175,3.505154639175,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1335","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",19.5876,19.5876,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1335--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",19.5876,19.5876,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1067","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",14.0206,14.0206,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1067--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",14.0206,14.0206,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:gpt-5-4:tau3-banking:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",39.587628865979404,39.587628865979404,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1050","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",30.3093,30.3093,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1050--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",30.3093,30.3093,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1285","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",21.4433,21.4433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1285--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",21.4433,21.4433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1148","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",21.0309,21.0309,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1148--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",21.0309,21.0309,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:gpt-5-5:tau3-banking:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",38.9690721649485,38.9690721649485,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-973","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.3402,31.3402,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-973--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.3402,31.3402,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:gpt-5-6-luna:tau3-banking:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.1340206185567,31.1340206185567,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-938","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",27.2165,27.2165,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-938--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",27.2165,27.2165,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:gpt-5-6-sol:tau3-banking:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",44.3298969072165,44.3298969072165,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-946","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",32.9897,32.9897,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-946--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",32.9897,32.9897,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:gpt-5-6-terra:tau3-banking:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",40.2061855670103,40.2061855670103,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-995","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.7526,31.7526,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-995--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.7526,31.7526,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1113","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",12.1649,12.1649,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:grok-4-5:tau3-banking:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",42.0618556701031,42.0618556701031,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1140","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",32.5773,32.5773,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:grok-4-6:tau3-banking:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",50.7216494845361,50.7216494845361,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1184","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",14.2268,14.2268,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:kimi-k2-5-reasoning:tau3-banking:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",14.226804123711,14.226804123711,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1413","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",20.6186,20.6186,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1431","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",18.1443,18.1443,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["aa-individual:kimi-k3:tau3-banking:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",45.979381443299,45.979381443299,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1950","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis τ³-Banking evaluation.",null,"Kimi K3; Artificial Analysis τ³-Banking evaluation.","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",33.4,33.4,"percent","higher","1.4.1","reference-only","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis τ³-Banking score for Kimi K3."],["aa-current:llama-4-maverick:tau3-banking:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",3.711340206186,3.711340206186,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:tau3-banking:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",3.298969072165,3.298969072165,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:tau3-banking:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.19587628866,13.19587628866,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["evidence-2026-07-1572","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",6.5979,6.5979,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-929","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",8.6598,8.6598,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1058","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",8.866,8.866,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1004","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",12.9897,12.9897,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1043","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.7732,5.7732,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1244","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",14.433,14.433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:mistral-medium-3-5-128b:tau3-banking:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",15.051546391753,15.051546391753,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-954","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.1546,5.1546,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:mistral-small-4-reasoning:tau3-banking:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",4.948453608247,4.948453608247,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1265","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",19.5876,19.5876,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-individual:muse-spark-1-1:tau3-banking:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",31.752577319587598,31.752577319587598,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-07-1237","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.1546,25.1546,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1237--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.1546,25.1546,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1924","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.2,25.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1924--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",25.2,25.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["aa-individual:muse-spark-1-2:tau3-banking:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",34.8453608247423,34.8453608247423,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual tau3-banking result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:tau3-banking:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.979381443299,5.979381443299,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1822","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",10.1031,10.1031,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1600","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.8144,13.8144,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1642","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",7.6289,7.6289,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-tau3-banking-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",9.28,9.28,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1496","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.6082,13.6082,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1478","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.4021,13.4021,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["aa-current:qwen3-5-397b-reasoning:tau3-banking:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",13.40206185567,13.40206185567,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:tau3-banking:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",15.257731958763,15.257731958763,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1456","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",15.2577,15.2577,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1272","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",16.4948,16.4948,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:qwen3-6-35b-a3b:tau3-banking:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",9.278350515464,9.278350515464,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-1124","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",10.9278,10.9278,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["evidence-2026-07-1202","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",17.8694,17.8694,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-tau3-banking","aa-tau3-banking","τ³-Banking Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/tau3-banking","2026-07-15","2026-07-15","2026-07-15","source-checked","τ³-Banking success rate from Artificial Analysis."],["aa-current:qwen-3-8-flash-next:tau3-banking:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",45.360824742268,45.360824742268,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:tau3-banking:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3",5.773195876289,5.773195876289,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-08-muse-glimmer-30b-tau3-banking-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3 / Banking",23.5,23.5,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["evidence-2026-08-muse-glimmer-30b-tau3-banking-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","3 / Banking",23.5,23.5,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["evidence-2026-08-15-command-a-plus-tau3-banking-banking-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",5.8,5.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-tau3-banking-banking-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",22.3,22.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-tau3-banking-banking-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",8.7,8.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-tau3-banking-banking-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",5.8,5.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-tau3-banking-banking-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",7.4,7.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-tau3-banking-banking","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","tau3-banking","τ³-Banking","agents","Sierra / Artificial Analysis","banking",19.6,19.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-opus-4-5-tau3bench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.2,17.8295,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-tau3bench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.2,17.8295,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-tau3bench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.2,17.8295,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau3bench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau3bench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.6,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-tau3bench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau3bench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.6,19.3798,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau3bench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.6,19.3798,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-tau3bench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.6,19.3798,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau3bench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.7,0.3876,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau3bench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.7,0.3876,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-tau3bench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",65.7,0.3876,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau3bench-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",72.9,28.2946,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau3bench-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",72.9,28.2946,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-tau3bench-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",72.9,28.2946,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",91.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",91.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",91.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau3bench-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.9,20.5426,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau3bench-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.9,20.5426,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-tau3bench-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.9,20.5426,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau3bench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",68.4,10.8527,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau3bench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",68.4,10.8527,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-tau3bench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",68.4,10.8527,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau3bench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.7,19.7674,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau3bench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.7,19.7674,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-tau3bench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",70.7,19.7674,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",67.2,6.2016,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",67.2,6.2016,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-tau3bench","τ³-Bench Tool-Agent-User Evaluation","agents","Sierra Research","2026",67.2,6.2016,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["xai-grok-4-6-release-apex-swe-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","apex-swe","APEX-SWE","coding","Mercor",null,58.8,58.8,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-apex-swe-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","apex-swe","APEX-SWE","coding","Mercor",null,53.6,53.6,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-apex-swe-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","apex-swe","APEX-SWE","coding","Mercor",null,56.4,56.4,"percent","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["benchlm-ref-claude-3-opus-aacodingindex-2026-07-21","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",19.53,16.4114,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aacodingindex-2026-07-27","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",19.53,17.3852,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aacodingindex-2026-08-01","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",19.53,17.3852,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aacodingindex-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.49,98.6998,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aacodingindex-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.49,97.894,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aacodingindex-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.49,97.894,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aacodingindex-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",73.6,94.5247,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aacodingindex-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",73.6,93.8092,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aacodingindex-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",73.6,93.8092,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aacodingindex-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.25,95.4637,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aacodingindex-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.25,94.7279,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aacodingindex-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.25,94.7279,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aacodingindex-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",77.98,100,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aacodingindex-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",77.98,100,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aacodingindex-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.55,91.5631,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aacodingindex-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.55,90.9117,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aacodingindex-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.55,90.9117,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aacodingindex-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",27.85,28.4311,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aacodingindex-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",27.85,29.1449,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aacodingindex-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",27.85,29.1449,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aacodingindex-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.04,21.4822,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aacodingindex-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.04,22.3463,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aacodingindex-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.04,22.3463,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2621,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2621,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2226,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2226,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2226,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aacodingindex-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",51.96,63.2226,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.3441,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.1731,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.1731,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.3441,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.1731,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.17,69.1731,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.9558,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.9558,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.7067,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.7067,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.7067,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aacodingindex-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.67,72.7067,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.9526,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.682,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.682,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.9526,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.682,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",59.36,73.682,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aacodingindex-2026-07-21","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.63,22.3346,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aacodingindex-2026-07-27","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.63,23.1802,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aacodingindex-2026-08-01","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",23.63,23.1802,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aacodingindex-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",33.25,36.2323,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aacodingindex-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",33.25,36.7774,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aacodingindex-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",33.25,36.7774,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.69,38.3126,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.69,38.8127,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.69,38.8127,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aacodingindex-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.83,87.6336,"index","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aacodingindex-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.83,87.0671,"index","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aacodingindex-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.83,87.0671,"index","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aacodingindex-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",70.14,89.5261,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aacodingindex-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",70.14,88.9187,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aacodingindex-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",70.14,88.9187,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.32,59.4481,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.32,59.4912,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.32,59.4912,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aacodingindex-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",69.24,88.2259,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aacodingindex-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",69.24,87.6466,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aacodingindex-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",69.24,87.6466,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aacodingindex-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",10.06,2.7304,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aacodingindex-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",10.06,4,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aacodingindex-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",10.06,4,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aacodingindex-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",30.96,33.5406,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aacodingindex-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",30.96,33.5406,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.32,45.0014,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.32,45.3569,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.32,45.3569,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aacodingindex-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",43.43,50.939,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aacodingindex-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",43.43,51.1661,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aacodingindex-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",43.43,51.1661,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aacodingindex-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",7.23,0,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aacodingindex-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",7.23,0,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aacodingindex-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",9.39,3.053,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aacodingindex-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",9.39,3.053,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aacodingindex-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.26,53.5828,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aacodingindex-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.26,53.7527,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aacodingindex-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.26,53.7527,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aacodingindex-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.78,68.7807,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aacodingindex-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.78,68.6219,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aacodingindex-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.78,68.6219,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aacodingindex-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.76,87.5325,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aacodingindex-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.76,86.9682,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aacodingindex-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",68.76,86.9682,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aacodingindex-2026-07-21","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",21.49,19.243,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aacodingindex-2026-07-27","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",21.49,20.1555,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aacodingindex-2026-08-01","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",21.49,20.1555,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aacodingindex-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.21,17.3938,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aacodingindex-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.21,18.3463,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aacodingindex-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.21,18.3463,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aacodingindex-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.14,4.2907,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aacodingindex-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.14,5.5265,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aacodingindex-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.14,5.5265,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aacodingindex-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.38,4.6374,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aacodingindex-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.38,5.8657,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aacodingindex-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",11.38,5.8657,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,42.7767,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,43.1802,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,43.1802,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,42.7767,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,43.1802,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aacodingindex-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",37.78,43.1802,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aacodingindex-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.39,59.5493,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aacodingindex-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.39,59.5901,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aacodingindex-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.39,59.5901,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aacodingindex-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.05,90.8408,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aacodingindex-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.05,90.2049,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aacodingindex-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.05,90.2049,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aacodingindex-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.08,69.2141,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aacodingindex-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.08,69.0459,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aacodingindex-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.08,69.0459,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aacodingindex-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.07,69.1997,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aacodingindex-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.07,69.0318,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aacodingindex-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",56.07,69.0318,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aacodingindex-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.89,96.3883,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aacodingindex-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.89,95.6325,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aacodingindex-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",74.89,95.6325,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aacodingindex-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.45,91.4187,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aacodingindex-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.45,90.7703,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aacodingindex-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.45,90.7703,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aacodingindex-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",77.39,100,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aacodingindex-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",77.39,99.1661,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aacodingindex-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",77.39,99.1661,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aacodingindex-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.66,98.9454,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aacodingindex-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.66,98.1343,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aacodingindex-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.66,98.1343,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aacodingindex-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",30.44,32.1728,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aacodingindex-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",30.44,32.8057,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aacodingindex-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",30.44,32.8057,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aacodingindex-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.7,18.1017,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aacodingindex-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.7,19.0389,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aacodingindex-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.7,19.0389,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aacodingindex-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",42.25,49.2343,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aacodingindex-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",42.25,49.4982,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aacodingindex-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",42.25,49.4982,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aacodingindex-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",72.45,92.8633,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aacodingindex-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",72.45,92.1837,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aacodingindex-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",72.45,92.1837,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aacodingindex-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,73.1436,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aacodingindex-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,72.8905,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aacodingindex-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,72.8905,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aacodingindex-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,73.1436,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aacodingindex-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,72.8905,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aacodingindex-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.8,72.8905,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aacodingindex-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.06,63.4065,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aacodingindex-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.06,63.364,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aacodingindex-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.06,63.364,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aacodingindex-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",32.11,35.1661,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aacodingindex-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",32.11,35.1661,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aacodingindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.7787,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aacodingindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.9011,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aacodingindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.9011,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aacodingindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.7787,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aacodingindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.9011,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aacodingindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.78,55.9011,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aacodingindex-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",61.77,77.4343,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aacodingindex-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",61.77,77.0883,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aacodingindex-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",61.77,77.0883,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aacodingindex-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.76,75.9752,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aacodingindex-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.76,75.6608,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aacodingindex-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.76,75.6608,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aacodingindex-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.24,98.3386,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aacodingindex-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.24,97.5406,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aacodingindex-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",76.24,97.5406,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aacodingindex-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.26,24.6894,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aacodingindex-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.26,25.4841,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aacodingindex-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.26,25.4841,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aacodingindex-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",16.28,11.7163,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aacodingindex-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",16.28,12.7915,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aacodingindex-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",16.28,12.7915,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aacodingindex-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",8.17,0,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aacodingindex-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",8.17,1.3286,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aacodingindex-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",8.17,1.3286,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aacodingindex-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.84,60.1994,"index","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aacodingindex-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.84,60.2261,"index","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aacodingindex-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.84,60.2261,"index","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.19,75.1517,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.19,74.8551,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",60.19,74.8551,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aacodingindex-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.62,64.2155,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aacodingindex-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.62,64.1555,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aacodingindex-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",52.62,64.1555,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aacodingindex-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.57,72.8113,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aacodingindex-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.57,72.5654,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aacodingindex-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.57,72.5654,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aacodingindex-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.07,17.1916,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aacodingindex-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.07,18.1484,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aacodingindex-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",20.07,18.1484,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.9,55.952,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.9,56.0707,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",46.9,56.0707,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aacodingindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,26.683,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aacodingindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,27.4346,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aacodingindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,27.4346,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aacodingindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,26.683,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aacodingindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,27.4346,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aacodingindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",26.64,27.4346,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aacodingindex-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.62,72.8836,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aacodingindex-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.62,72.636,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aacodingindex-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",58.62,72.636,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aacodingindex-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.34,91.2598,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aacodingindex-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.34,90.6148,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aacodingindex-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",71.34,90.6148,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",14.37,8.9569,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",14.37,10.0919,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",14.37,10.0919,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",13.75,8.0613,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",13.75,9.2155,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",13.75,9.2155,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aacodingindex-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.27,59.3759,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aacodingindex-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.27,59.4205,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aacodingindex-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",49.27,59.4205,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aacodingindex-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.72,45.5793,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aacodingindex-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.72,45.9223,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aacodingindex-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.72,45.9223,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-preview-aacodingindex-2026-07-21","o1-preview","o1-preview","Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.05,37.388,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-preview-aacodingindex-2026-07-27","o1-preview","o1-preview","Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.05,37.9081,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-preview-aacodingindex-2026-08-01","o1-preview","o1-preview","Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",34.05,37.9081,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aacodingindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.8446,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aacodingindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.9223,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aacodingindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.9223,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aacodingindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.8446,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aacodingindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.9223,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aacodingindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",48.21,57.9223,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.71,54.2329,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.71,54.3887,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",45.71,54.3887,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aacodingindex-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",53.72,65.8047,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aacodingindex-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",53.72,65.7102,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aacodingindex-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",53.72,65.7102,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aacodingindex-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",54.53,66.9749,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aacodingindex-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",54.53,66.8551,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aacodingindex-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",54.53,66.8551,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",41.88,48.6998,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",41.88,48.9753,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",41.88,48.9753,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aacodingindex-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",65.97,83.5019,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aacodingindex-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",65.97,83.0247,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aacodingindex-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",65.97,83.0247,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aacodingindex-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.86,68.8963,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aacodingindex-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.86,68.735,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aacodingindex-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",55.86,68.735,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aacodingindex-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.57,45.3626,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aacodingindex-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.57,45.7102,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aacodingindex-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",39.57,45.7102,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aacodingindex-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.77,26.2049,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aacodingindex-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.77,26.2049,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aacodingindex-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.77,26.2049,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aacodingindex-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aacodingindex","Artificial Analysis Coding Index","coding","Artificial Analysis","2026",25.77,26.2049,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aalivecodebench-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",91.7,100,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aalivecodebench-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",91.7,100,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aalivecodebench-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",91.7,100,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aalivecodebench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",89.4,41.0256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aalivecodebench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",89.4,41.0256,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aalivecodebench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",89.4,41.0256,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aalivecodebench-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",87.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aalivecodebench-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",87.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aalivecodebench-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aalivecodebench","Artificial Analysis LiveCodeBench","coding","Artificial Analysis","2026",87.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaterminalbench21-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,85.8333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaterminalbench21-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,82.0717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaterminalbench21-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,86.7647,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,85.8333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,82.0717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.6,86.7647,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaterminalbench21-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",89.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaterminalbench21-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",89.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.5,68.75,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.5,65.7371,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.5,74.7059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaterminalbench21-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",61.8,19.7059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaterminalbench21-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",61.8,19.7059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,26.1765,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",64,26.1765,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaterminalbench21-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",73.8,40.8333,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaterminalbench21-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",73.8,39.0438,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.5,56.25,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.5,53.7849,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.5,65.8824,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaterminalbench21-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,57.9167,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaterminalbench21-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,55.3785,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaterminalbench21-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,67.0588,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaterminalbench21-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.3,84.5833,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaterminalbench21-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",84.3,80.8765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.9,70.4167,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.9,67.3307,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",80.9,75.8824,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,95.6175,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,96.7647,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,95.6175,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",88,96.7647,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaterminalbench21-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",81.6,73.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaterminalbench21-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",81.6,70.1195,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaterminalbench21-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",81.6,77.9412,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaterminalbench21-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",55.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaterminalbench21-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",85,87.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaterminalbench21-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",85,83.6653,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaterminalbench21-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",85,87.9412,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,4.7809,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,29.7059,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaterminalbench21-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaterminalbench21-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,4.7809,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaterminalbench21-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",65.2,29.7059,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,57.9167,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,55.3785,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",77.9,67.0588,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaterminalbench21-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",74.5,43.75,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaterminalbench21-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",74.5,41.8327,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaterminalbench21-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-aaterminalbench21","Artificial Analysis Terminal-Bench v2.1","coding","Artificial Analysis","2026",74.5,57.0588,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-394","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",81.7,81.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-389","claude-mythos-5","Claude Mythos 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",81.95,81.95,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-521","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",58.72,58.72,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-454","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",65.66,65.66,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-438","claude-opus-4-7","Claude Opus 4.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",69.36,69.36,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-404","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",73.49,73.49,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-509","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",59.97,59.97,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-495","claude-sonnet-5","Claude Sonnet 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",64.66,64.66,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-554","deepseek-v4-pro","DeepSeek V4 Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",59.28,59.28,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-548","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",45.94,45.94,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-467","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",63.85,63.85,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-421","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",65.4,65.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-543","gemma-4-31b","Gemma 4 31B","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",50.09,50.09,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-489","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",60.28,60.28,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-460","glm-5-1","GLM-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",56.67,56.67,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-531","glm-5-2","GLM-5.2","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",62.31,62.31,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-534","glm-5v-turbo","GLM-5V-Turbo","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",59.85,59.85,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-515","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",51.03,51.03,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-484","gpt-5-3-codex","GPT-5.3-Codex","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",64.08,64.08,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-427","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",64.17,64.17,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-478","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",51.6,51.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-415","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",71.38,71.38,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-473","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",64.26,64.26,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-399","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",74.08,74.08,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-433","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",66.59,66.59,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-499","grok-4-3","Grok 4.3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",40.23,40.23,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-413","grok-4-5","Grok 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",71.25,71.25,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-464","mimo-v2-pro","MiMo-V2-Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",63.63,63.63,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-447","mimo-v2-5-pro","MiMo-V2.5-Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",55.16,55.16,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-527","minimax-m2-7","MiniMax M2.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",44.74,44.74,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-450","minimax-m3","MiniMax M3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",58.06,58.06,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-441","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",66.36,66.36,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-409","muse-spark-1-1","Muse Spark 1.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",69.1,69.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-503","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-coding","BenchLM Coding prior","coding","BenchLM","bench-align-v5.1",60.55,60.55,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (coding) used to estimate missing category coverage under methodology 1.3.0."],["benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",56.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",56.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",56.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",59.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",59.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","bigcodebench","BigCodeBench","coding","BigCodeBench authors","2026",59.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-code-arena-webdev-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","code-arena-webdev","Code Arena Web Development","coding","Code Arena","August 2026",1541,54.1,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-code-arena-webdev-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","code-arena-webdev","Code Arena Web Development","coding","Code Arena","August 2026",1538,53.8,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-code-arena-webdev-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","code-arena-webdev","Code Arena Web Development","coding","Code Arena","August 2026",1588,58.8,"Elo","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-code-arena-webdev-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","code-arena-webdev","Code Arena Web Development","coding","Code Arena","August 2026",1523,52.3,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-code-arena-webdev-muse-spark-1-2-2026-08-13","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2","muse-spark-1-2-xhigh","Muse Spark 1.2","code-arena-webdev","Code Arena Web Development","coding","Code Arena","August 2026",1535,53.5,"Elo","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-codeforces-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2816,0,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-codeforces-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3052,60.5128,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-codeforces-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",2919,26.4103,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-codeforces-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-codeforces","Codeforces Rating","coding","DeepSeek-AI","2026",3206,100,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-378","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",70.5,70.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-735","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",70.5,70.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["xai-grok-4-6-release-cursorbench-v3-2-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","cursor-bench","CursorBench","coding","Cursor","3.2",70.5,70.5,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-382","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",62.3,62.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-739","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",62.3,62.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-383","claude-sonnet-5","Claude Sonnet 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",61.5,61.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-740","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",61.5,61.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-387","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",48.8,48.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-744","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",48.8,48.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-386","glm-5-2","GLM-5.2","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",55,55,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-743","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",55,55,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-385","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",58.4,58.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-742","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",58.4,58.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-384","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",61.1,61.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-741","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",61.1,61.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-379","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",67.2,67.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-736","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",67.2,67.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["xai-grok-4-6-release-cursorbench-v3-2-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","cursor-bench","CursorBench","coding","Cursor","3.2",67.2,67.2,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["evidence-2026-07-381","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",64.9,64.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-738","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",64.9,64.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-380","grok-4-5","Grok 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","cursor-bench","CursorBench","coding","Cursor","3.2",66.7,66.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::cursor-bench-public","cursor-bench-public","CursorBench public evaluation via BenchLM","Cursor / BenchLM","https://benchlm.ai/benchmarks/cursorBench","2026-07-15","2026-07-15","2026-07-15","source-checked","CursorBench 3.2 public score via BenchLM."],["evidence-2026-07-737","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","cursor-bench","CursorBench","coding","Cursor","3.2",66.7,66.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["xai-grok-4-6-release-cursorbench-v3-2-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","cursor-bench","CursorBench","coding","Cursor","3.2",66.7,66.7,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-cursorbench-v3-2-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","cursor-bench","CursorBench","coding","Cursor","3.2",69.9,69.9,"percent","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["new-model:cursorbench-fable-5-1-2026-09-01:claude-fable-5-1:cursor-bench:cursorbench-3.2.0:claude-fable-5-1:max","claude-fable-5-1","Claude Fable 5.1","Claude Fable 5.1 Max on CursorBench 3.2.0","claude-fable-5-1-max","Claude Fable 5.1 Max on CursorBench 3.2.0","cursor-bench","CursorBench","coding","Cursor","3.2.0",73.4,73.4,"percent","higher","2.1.0","reference-only","direct","2026-09-01",null,"production::cursorbench-fable-5-1-2026-09-01","cursorbench-fable-5-1-2026-09-01","CursorBench 3.2.0 cost savings benchmark table","Cursor","https://cursor.com/en-US/cost-savings","2026-09-01","2026-09-01","2026-09-01","source-checked","First-party Cursor result. Reference-only because CursorBench is not weighted by Methodology 2.2. The matching Anthropic launch-table carrier is the same underlying measurement, not a second independent result."],["benchlm-ref-deepseek-v4-flash-max-dsbenchfullstack-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","2026",68.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-dsbenchfullstack-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","2026",68.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-dsbench-fullstack-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",77.2,77.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",71.6,71.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",37,37,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",68.7,68.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",41.8,41.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",71.1,71.1,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",61.8,61.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-fullstack-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-dsbenchfullstack","DeepSeek DSBench FullStack","coding","DeepSeek-AI","August 2026",73.7,73.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["benchlm-ref-deepseek-v4-flash-max-dsbenchhard-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","2026",59.6,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-dsbenchhard-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","2026",59.6,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-dsbench-hard-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",68.3,68.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",71.7,71.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",25.8,25.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",59.6,59.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-flash-vision-exp-dsbench-hard-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",63.6,63.6,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value. DeepSeek identifies DSBench-Hard as an internal Coding Agent test set; it remains reference-only because no independent benchmark-owner protocol exists."],["deepseek-v4-pro-0813-release-dsbench-hard-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",31.1,31.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",67.2,67.2,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",54.5,54.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-dsbench-hard-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-dsbenchhard","DeepSeek DSBench Hard","coding","DeepSeek-AI","August 2026",63,63,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-08-15-claude-fable-5-deepswe-1-1-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",69.7,69.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-15-claude-opus-4-8-deepswe-1-1-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",58,58,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:deepswe:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepswe","DeepSWE","coding","DataCurve","1.1",54.4,54.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","DeepSWE 1.1 reports the highest result across Claude Code and mini-SWE-agent. Qwen3.8-Flash-Next's best result uses mini-SWE-agent with temperature 1, top_p 0.95 and a 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-deepseek-v4-pro-0813-deepswe-1-1-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",62.7,62.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-15-glm-5-2-deepswe-1-1-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",46.2,46.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-15-glm-5-3-deepswe-1-1","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","deepswe","DeepSWE","coding","DataCurve","1.1",66.9,66.9,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-15-gpt-5-6-sol-deepswe-1-1-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",72.7,72.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-15-kimi-k3-deepswe-1-1-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",67.5,67.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["evidence-2026-08-muse-spark-1-2-deepswe-1-1","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) with Muse Code",null,"Muse Spark 1.2 (xhigh) with Muse Code","deepswe","DeepSWE","coding","DataCurve","1.1",59.3,59.3,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 59.3% over five attempts. The methodology says Pier 0.3.0 was attempted but the published score uses Muse Code, so this row is not harness-identical to the official DeepSWE leaderboard."],["evidence-2026-08-muse-spark-1-2-deepswe-1-1--configuration--muse-spark-1-2-xhigh","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) with Muse Code","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh) with Muse Code","deepswe","DeepSWE","coding","DataCurve","1.1",59.3,59.3,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 59.3% over five attempts. The methodology says Pier 0.3.0 was attempted but the published score uses Muse Code, so this row is not harness-identical to the official DeepSWE leaderboard."],["evidence-2026-08-15-qwen3-6-27b-deepswe-1-1-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","deepswe","DeepSWE","coding","DataCurve","1.1",13.3,13.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:deepswe:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepswe","DeepSWE","coding","DataCurve","1.1",16.5,16.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","DeepSWE 1.1 reports the highest result across Claude Code and mini-SWE-agent. Qwen3.8-Flash-Next's best result uses mini-SWE-agent with temperature 1, top_p 0.95 and a 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-deepswe-1-1-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","deepswe","DeepSWE","coding","DataCurve","1.1",14.2,14.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-qwen-3-8-max-deepswe-1-1","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","deepswe","DeepSWE","coding","DataCurve","1.1",56.6,56.6,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 56.6. Qwen evaluated Claude Code and mini-SWE-agent and published the higher harness result; the best score was obtained with Claude Code."],["evidence-2026-08-qwen-3-8-max-deepswe-1-1--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","deepswe","DeepSWE","coding","DataCurve","1.1",56.6,56.6,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 56.6. Qwen evaluated Claude Code and mini-SWE-agent and published the higher harness result; the best score was obtained with Claude Code."],["evidence-2026-08-15-qwen-3-8-max-deepswe-1-1-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","deepswe","DeepSWE","coding","DataCurve","1.1",56.6,56.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: mini-swe-agent, temp=0.95, top_p=1.0, timeout=6h, 400K context."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:deepswe:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepswe","DeepSWE","coding","DataCurve","1.1",42.2,42.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","DeepSWE 1.1 reports the highest result across Claude Code and mini-SWE-agent. Qwen3.8-Flash-Next's best result uses mini-SWE-agent with temperature 1, top_p 0.95 and a 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-deepswe-1-1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","deepswe","DeepSWE","coding","DataCurve","1.1",42.2,42.2,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-opus-5-deepswe-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",68.8,87.9257,"USD","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-deepswe-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",68.8,87.9257,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-deepswe-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",54.4,43.3437,"USD","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-deepswe-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",54.4,43.3437,"USD","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-deepswe-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",49,26.6254,"USD","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-deepswe-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",49,26.6254,"USD","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-deepswe-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",49,26.6254,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-deepswe-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.2,82.9721,"USD","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-deepswe-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.2,82.9721,"USD","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-deepswe-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.2,82.9721,"USD","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-deepswe-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",72.7,100,"USD","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-deepswe-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",72.7,100,"USD","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-deepswe-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",72.7,100,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-deepswe-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",69.6,90.4025,"USD","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-deepswe-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",69.6,90.4025,"USD","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-deepswe-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",69.6,90.4025,"USD","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-deepswe-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53,39.0093,"USD","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-deepswe-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53,39.0093,"USD","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-deepswe-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53,39.0093,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deepswe-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.5,83.9009,"USD","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deepswe-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.5,83.9009,"USD","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-deepswe-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",67.5,83.9009,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-deepswe-2026-07-21","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",40.4,0,"USD","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-deepswe-2026-07-27","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",40.4,0,"USD","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-deepswe-2026-08-01","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",40.4,0,"USD","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepswe-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53.3,39.9381,"USD","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepswe-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53.3,39.9381,"USD","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-deepswe-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","deepswe","DeepSWE","coding","DataCurve","2026",53.3,39.9381,"USD","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-deepswe-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","deepswe","DeepSWE","coding","DataCurve","v1.1",70,70,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["xai-grok-4-6-release-deepswe-v1-1-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","deepswe","DeepSWE","coding","DataCurve","v1.1",70,70,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["deepseek-v4-pro-0813-release-deepswe-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","deepswe","DeepSWE","coding","DataCurve","v1.1",58,58,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-claude-opus-5","claude-opus-5","Claude Opus 5","Claude Opus 5 (reasoning configuration not stated in chart)","claude-opus-5-meta-muse-12-release-unspecified","Claude Opus 5 (reasoning configuration not stated in chart)","deepswe","DeepSWE","coding","DataCurve","v1.1",65,65,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-2063","claude-opus-5","Claude Opus 5","DeepSWE v1.1 agentic coding as published in the Anthropic Claude Opus 5 launch table.",null,"DeepSWE v1.1 agentic coding as published in the Anthropic Claude Opus 5 launch table.","deepswe","DeepSWE","coding","DataCurve","v1.1",68.8,68.8,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 68.8% on DeepSWE v1.1."],["evidence-2026-07-2028","claude-sonnet-5","Claude Sonnet 5","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","deepswe","DeepSWE","coding","DataCurve","v1.1",54,54,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 DeepSWE comparison score."],["google-gemini-37-eval-deepswe-v1-1-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","deepswe","DeepSWE","coding","DataCurve","v1.1",53.8,53.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["deepseek-v4-pro-0813-release-deepswe-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","deepswe","DeepSWE","coding","DataCurve","v1.1",7.3,7.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-flash-0731-deepswe-v1-1-2026-07-31","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-0731-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepswe","DeepSWE","coding","DataCurve","v1.1",54.4,54.4,"percent","higher","2.3.0","reference-only","direct","2026-07-31","2026-07-31","production::deepseek-v4-flash-0731-model-card","deepseek-v4-flash-0731-model-card","DeepSeek-V4-Flash-0731 official model card and provider evaluation","DeepSeek","https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","2026-07-31","2026-08-24","2026-08-24","provider-reported","Source-native DeepSeek subject value with the official code-agent configuration; the exact DeepSWE agent scaffold and track remain unresolved."],["deepseek-v4-pro-0813-release-deepswe-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","deepswe","DeepSWE","coding","DataCurve","v1.1",54.4,54.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-flash-vision-exp-deepswe-v1-1-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepswe","DeepSWE","coding","DataCurve","v1.1",59.3,59.3,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value. DeepSWE v1.1 identity and metric are resolved, but the source does not resolve the required agent scaffold and track."],["deepseek-v4-pro-0813-release-deepswe-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","deepswe","DeepSWE","coding","DataCurve","v1.1",12.8,12.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-deepswe-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","deepswe","DeepSWE","coding","DataCurve","v1.1",62.7,62.7,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["evidence-2026-07-2009","gemini-3-5-flash","Gemini 3.5 Flash","As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.",null,"As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.","deepswe","DeepSWE","coding","DataCurve","v1.1",37,37,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","DeepSWE comparison row for 3.5 Flash from 3.6 launch materials."],["google-gemini-37-eval-deepswe-v1-1-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","deepswe","DeepSWE","coding","DataCurve","v1.1",48.6,48.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-gemini-3-6-flash","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (reasoning configuration not stated in chart)","gemini-3-6-flash-meta-muse-12-release-unspecified","Gemini 3.6 Flash (reasoning configuration not stated in chart)","deepswe","DeepSWE","coding","DataCurve","v1.1",40,40,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-1989","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). High reasoning; public DeepSWE v1.1 leaderboard as cited by Google.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). High reasoning; public DeepSWE v1.1 leaderboard as cited by Google.","deepswe","DeepSWE","coding","DataCurve","v1.1",49,49,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","DeepSWE 49% (vs 37% for 3.5 Flash) cited in Google launch blog; harness-dependent."],["evidence-2026-07-1989--configuration--gemini-3-6-flash-high","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). High reasoning; public DeepSWE v1.1 leaderboard as cited by Google.","gemini-3-6-flash-high","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). High reasoning; public DeepSWE v1.1 leaderboard as cited by Google.","deepswe","DeepSWE","coding","DataCurve","v1.1",49,49,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","DeepSWE 49% (vs 37% for 3.5 Flash) cited in Google launch blog; harness-dependent."],["google-gemini-37-eval-deepswe-v1-1-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-high","Gemini 3.7 Flash","deepswe","DeepSWE","coding","DataCurve","v1.1",65.3,65.3,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["deepseek-v4-pro-0813-release-deepswe-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","deepswe","DeepSWE","coding","DataCurve","v1.1",46.2,46.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-07-2015","gpt-5-6-luna","GPT-5.6 Luna","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","deepswe","DeepSWE","coding","DataCurve","v1.1",67,67,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna DeepSWE comparison score."],["xai-grok-4-6-release-deepswe-v1-1-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","deepswe","DeepSWE","coding","DataCurve","v1.1",73,73,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["google-gemini-37-eval-deepswe-v1-1-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","deepswe","DeepSWE","coding","DataCurve","v1.1",69.6,69.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-gpt-5-6-terra","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (reasoning configuration not stated in chart)","gpt-5-6-terra-meta-muse-12-release-unspecified","GPT-5.6 Terra (reasoning configuration not stated in chart)","deepswe","DeepSWE","coding","DataCurve","v1.1",64.8,64.8,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-07-2022","grok-4-5","Grok 4.5","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","deepswe","DeepSWE","coding","DataCurve","v1.1",54,54,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Grok 4.5 DeepSWE comparison score."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-grok-4-5","grok-4-5","Grok 4.5","Grok 4.5 (reasoning configuration not stated in chart)","grok-4-5-meta-muse-12-release-unspecified","Grok 4.5 (reasoning configuration not stated in chart)","deepswe","DeepSWE","coding","DataCurve","v1.1",56.6,56.6,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["xai-grok-4-6-release-deepswe-v1-1-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","deepswe","DeepSWE","coding","DataCurve","v1.1",54,54,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-deepswe-v1-1-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","deepswe","DeepSWE","coding","DataCurve","v1.1",65.9,65.9,"percent","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["deepseek-v4-pro-0813-release-deepswe-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","deepswe","DeepSWE","coding","DataCurve","v1.1",67.5,67.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["evidence-2026-07-1956","kimi-k3","Kimi K3","mini-SWE-agent common harness; DeepSWE v1.1; max reasoning. Provider also reports 67.5 with KimiCode.",null,"mini-SWE-agent common harness; DeepSWE v1.1; max reasoning. Provider also reports 67.5 with KimiCode.","deepswe","DeepSWE","coding","DataCurve","v1.1",67.3,67.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Prefer common-harness 67.3 (mini-SWE-agent) over KimiCode 67.5 for comparability. Official board listing pending."],["evidence-2026-07-1956--configuration--kimi-k3-max","kimi-k3","Kimi K3","mini-SWE-agent common harness; DeepSWE v1.1; max reasoning. Provider also reports 67.5 with KimiCode.","kimi-k3-max","mini-SWE-agent common harness; DeepSWE v1.1; max reasoning. Provider also reports 67.5 with KimiCode.","deepswe","DeepSWE","coding","DataCurve","v1.1",67.3,67.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Prefer common-harness 67.3 (mini-SWE-agent) over KimiCode 67.5 for comparability. Official board listing pending."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-muse-spark-1-1","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (reasoning configuration not stated in chart)","muse-spark-1-1-meta-muse-12-release-unspecified","Muse Spark 1.1 (reasoning configuration not stated in chart)","deepswe","DeepSWE","coding","DataCurve","v1.1",53,53,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["google-gemini-37-eval-deepswe-v1-1-muse-spark-1-2-2026-08-13","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2","muse-spark-1-2-xhigh","Muse Spark 1.2","deepswe","DeepSWE","coding","DataCurve","v1.1",54.9,54.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-meta-muse-12-deepswe-1-1-muse-spark-1-2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","deepswe","DeepSWE","coding","DataCurve","v1.1",59.3,59.3,"percent","higher","2.0.0","reference-only","direct","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-deepswe-chart","refresh-meta-muse-spark-1-2-deepswe-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/deepswe-1-1-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Exact Muse Spark 1.2 result retained with the published xhigh setting."],["benchlm-ref-deepseek-v4-flash-base-humaneval-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",69.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-humaneval-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",69.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-humaneval-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",69.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-humaneval-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",76.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-humaneval-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",76.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-humaneval-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",76.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-humaneval-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",73.8,58.9041,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-humaneval-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",73.8,58.9041,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-humaneval-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-humaneval","Evaluating Large Language Models Trained on Code","coding","Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","2021",73.8,58.9041,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["xai-grok-4-6-release-frontiercode-v1-1-extended-claude-fable-5-max-2026-08-12","claude-fable-5","Claude Fable 5","Fable 5 Max","claude-fable-5-max","Fable 5 Max","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","1.1 Extended",63.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-frontiercode-v1-1-extended-gpt-5-6-sol-max-2026-08-12","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol Max","gpt-5-6-sol-max","GPT-5.6 Sol Max","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","1.1 Extended",60.6,57.142857,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-frontiercode-v1-1-extended-grok-4-5-high-2026-08-12","grok-4-5","Grok 4.5","Grok 4.5 High","grok-4-5-aa-2-high","Grok 4.5 High","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","1.1 Extended",56.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","xAI labels third-party model values as the best self-reported or publicly available result."],["xai-grok-4-6-release-frontiercode-v1-1-extended-grok-4-6-high-2026-08-12","grok-4-6","Grok 4.6","Grok 4.6 High","grok-4-6-high","Grok 4.6 High","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","1.1 Extended",61.3,67.142857,"percent","higher","1.8.0","reference-only","direct","2026-08-12","2026-08-12","production::xai-grok-4-6-release-2026-08-12","xai-grok-4-6-release-2026-08-12","Grok 4.6","xAI","https://x.ai/news/grok-4-6","2026-08-12","2026-08-12","2026-08-12","provider-reported","Official xAI provider result."],["benchlm-ref-claude-opus-5-frontiercode11extended-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",63.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-frontiercode11extended-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",63.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",60.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",60.6,64.7059,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",60.6,64.7059,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.8,12.7273,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.8,8.2353,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode11extended","FrontierCode 1.1 Extended","coding","Cognition","2026",55.8,8.2353,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2064","claude-opus-5","Claude Opus 5","FrontierCode v1.1 Main as published in the Anthropic Claude Opus 5 launch table.",null,"FrontierCode v1.1 Main as published in the Anthropic Claude Opus 5 launch table.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","1.1",53.4,53.4,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 53.4% on FrontierCode v1.1 Main (matches Fable 5 at 53.5% in the same table)."],["google-gemini-37-eval-frontiercode-1-1-main-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","1.1 Main",42.7,42.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-frontiercode-1-1-main-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","1.1 Main",34.4,34.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-frontiercode-1-1-main-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","1.1 Main",43.6,43.6,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-frontiercode-1-1-main-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","1.1 Main",41.3,41.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-claude-fable-frontiercode-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",53.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-frontiercode-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",53.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-frontiercode-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",53.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiercode-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",26.9,8.9041,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiercode-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",26.9,8.9041,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiercode-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",26.9,8.9041,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiercode-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",38.5,48.6301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiercode-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",38.5,48.6301,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiercode-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",38.5,48.6301,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiercode-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",46.5,76.0274,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiercode-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",46.5,76.0274,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiercode-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",46.5,76.0274,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-frontiercode-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",53.4,99.6575,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-frontiercode-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",53.4,99.6575,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiercode-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",24.3,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiercode-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",24.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiercode-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",24.3,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-frontiercode-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.7,63.0137,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-frontiercode-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.7,63.0137,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-frontiercode-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.7,63.0137,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiercode-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",27,9.2466,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiercode-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",27,9.2466,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiercode-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",27,9.2466,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiercode-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",43,64.0411,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiercode-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",43,64.0411,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiercode-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",43,64.0411,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-frontiercode-2026-07-21","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.3,61.6438,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-frontiercode-2026-07-27","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.3,61.6438,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-frontiercode-2026-08-01","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiercode","FrontierCode 1.1 Main","coding","Cognition","2026",42.3,61.6438,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-frontierswe-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","frontierswe","FrontierSWE","coding","FrontierSWE","2026",81.2,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-frontierswe-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","frontierswe","FrontierSWE","coding","FrontierSWE","2026",81.2,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-frontierswe-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","frontierswe","FrontierSWE","coding","FrontierSWE","2026",81.2,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-fable-5-frontierswe-2026-07-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",88.2,88.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-opus-4-8-frontierswe-2026-07-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",66.5,66.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-2-frontierswe-2026-07-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",67.5,67.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-3-frontierswe-2026-07","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",78.1,78.1,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-1959","kimi-k3","Kimi K3","KimiCode harness; FrontierSWE dominance recomputed with official script as of 2026-07-16.",null,"KimiCode harness; FrontierSWE dominance recomputed with official script as of 2026-07-16.","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",81.2,81.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Dominance score 81.2; official FrontierSWE board had not yet listed K3 at check date."],["evidence-2026-07-1959--configuration--kimi-k3-max","kimi-k3","Kimi K3","KimiCode harness; FrontierSWE dominance recomputed with official script as of 2026-07-16.","kimi-k3-max","KimiCode harness; FrontierSWE dominance recomputed with official script as of 2026-07-16.","frontierswe","FrontierSWE","coding","FrontierSWE","2026-07",81.2,81.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Dominance score 81.2; official FrontierSWE board had not yet listed K3 at check date."],["benchlm-ref-minimax-m3-kernelbenchhard-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-kernelbenchhard","KernelBench Hard","coding","MiniMax","2026",28.8,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-kernelbenchhard-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-kernelbenchhard","KernelBench Hard","coding","MiniMax","2026",28.8,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-kernelbenchhard-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-kernelbenchhard","KernelBench Hard","coding","MiniMax","2026",28.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1961","kimi-k3","Kimi K3","KimiCode and Claude Code harnesses at max effort; internal Kimi Code Bench 2.0.",null,"KimiCode and Claude Code harnesses at max effort; internal Kimi Code Bench 2.0.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2.0",72.9,72.9,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Internal Moonshot benchmark; reference-only, not used in ranking weight."],["evidence-2026-07-1961--configuration--kimi-k3-max","kimi-k3","Kimi K3","KimiCode and Claude Code harnesses at max effort; internal Kimi Code Bench 2.0.","kimi-k3-max","KimiCode and Claude Code harnesses at max effort; internal Kimi Code Bench 2.0.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2.0",72.9,72.9,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Internal Moonshot benchmark; reference-only, not used in ranking weight."],["benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",62,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",62,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",62,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-kimicodebenchv2-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",72.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-kimicodebenchv2-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",72.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-kimicodebenchv2-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","kimi-code-bench-v2","Kimi Code Bench v2","coding","Moonshot AI","2026",72.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-livecodebench-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",37.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-livecodebench-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",37.6,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-livecodebench-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",37.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-livecodebench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",84.9,87.5926,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-livecodebench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",84.9,87.5926,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-livecodebench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",84.9,87.5926,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-livecodebench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",83.9,85.7407,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-livecodebench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",83.9,85.7407,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-livecodebench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",83.9,85.7407,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",80.4,79.2593,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",80.4,79.2593,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",80.4,79.2593,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-livecodebench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",91.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-livecodebench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",91.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-livecodebench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",91.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-livecodebench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",89.6,96.2963,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-livecodebench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",89.6,96.2963,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-livecodebench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","livecodebench","LiveCodeBench","coding","LiveCodeBench team","2024",89.6,96.2963,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["livecodebench-lcb:claude-3-haiku:1722470400000-1746057600000","claude-3-haiku","Claude 3 Haiku","Claude-3-Haiku",null,"Claude-3-Haiku","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",20.2,20.2,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=62.3; medium=12.6; hard=2.8; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:claude-3-5-sonnet-20241022:1722470400000-1746057600000","claude-35-sonnet","Claude 3.5 Sonnet (Oct '24)","Claude-3.5-Sonnet-20241022",null,"Claude-3.5-Sonnet-20241022","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",36.4,36.4,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=91.2; medium=34.3; hard=8.2; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:claude-opus-4:1722470400000-1746057600000","claude-opus-4","Claude Opus 4","Claude-Opus-4",null,"Claude-Opus-4","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",46.9,46.9,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=94.5; medium=52.5; hard=17.2; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:claude-sonnet-4:1722470400000-1746057600000","claude-sonnet-4","Claude Sonnet 4","Claude-Sonnet-4",null,"Claude-Sonnet-4","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",47.1,47.1,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=96.4; medium=53.9; hard=15.8; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:deepseek-r1-0528:1722470400000-1746057600000","deepseek-r1-aa-2","DeepSeek R1 0528 (May '25)","DeepSeek-R1-0528",null,"DeepSeek-R1-0528","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",73.1,73.1,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=98.7; medium=85.2; hard=50.7; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:deepseek-v3:1722470400000-1746057600000","deepseek-v3","DeepSeek V3","DeepSeek-V3",null,"DeepSeek-V3","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",27.2,27.2,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=64.3; medium=27.9; hard=6.7; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:exaone-4-0-32b:1722470400000-1746057600000","exaone-4-0-32b","Exaone 4.0 32B","EXAONE-4.0-32B",null,"EXAONE-4.0-32B","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",70,70,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=98.4; medium=82.3; hard=46.2; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:gemini-2-5-pro-05-06:1722470400000-1746057600000","gemini-2-5-pro-05-06","Gemini 2.5 Pro Preview (May' 25)","Gemini-2.5-Pro-05-06",null,"Gemini-2.5-Pro-05-06","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",71.8,71.8,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=98.2; medium=82.3; hard=50.2; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:gpt-4-turbo-2024-04-09:1722470400000-1746057600000","gpt-4-turbo","GPT-4 Turbo","GPT-4-Turbo-2024-04-09",null,"GPT-4-Turbo-2024-04-09","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",28.7,28.7,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=81; medium=24.8; hard=3.1; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:gpt-4o-2024-08-06:1722470400000-1746057600000","gpt-4o-2024-08-06","GPT-4o (Aug '24)","GPT-4O-2024-08-06",null,"GPT-4O-2024-08-06","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",29.5,29.5,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=82.5; medium=26; hard=3.3; contaminated=false; start=2024-08-01; end=2025-05-01."],["livecodebench-lcb:gpt-4o-mini-2024-07-18:1722470400000-1746057600000","gpt-4o-mini","GPT-4o mini","GPT-4O-mini-2024-07-18",null,"GPT-4O-mini-2024-07-18","livecodebench","LiveCodeBench","coding","LiveCodeBench team","standard",27.5,27.5,"percent","higher","2.2.0","ranking-eligible","direct","2025-05-01","2025-05-01","production::refresh-livecodebench","refresh-livecodebench","LiveCodeBench permanent refresh source","LiveCodeBench","https://livecodebench.github.io/leaderboard.html","2025-05-01","2026-08-07","2026-08-18","official-leaderboard","Official LiveCodeBench default window Pass@1; n=454; easy=81.9; medium=18.9; hard=3.9; contaminated=false; start=2024-08-01; end=2025-05-01."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:lcb-v6:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",88.8,88.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-livecodebench-v6-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",88.8,88.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-command-a-plus-livecodebench-v6-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",86.1,86.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-livecodebench-v6-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",92.3,92.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:lcb-v6:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",90.6,90.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-mimo-v2-5-livecodebench-v6-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",89.1,89.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-livecodebench-v6-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",84.9,84.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-livecodebench-v6-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",83.9,83.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["new-model:qwen3-6-35b-a3b-model-card-2026-08-29:qwen3-6-35b-a3b:livecodebench:cell:qwen3-6-35b-a3b:livecodebench-v6","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6-35B-A3B default thinking configuration","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6-35B-A3B default thinking configuration","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",80.4,80.4,"percent","higher","2.1.0","reference-only","direct","2026-04-15",null,"production::qwen3-6-35b-a3b-model-card-2026-08-29","qwen3-6-35b-a3b-model-card-2026-08-29","Qwen3.6-35B-A3B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.6-35B-A3B","2026-04-15","2026-08-29","2026-08-29","source-checked","Provider-direct result; reference-only because LiveCodeBench v6 is not scoring-active."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:lcb-v6:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",89.6,89.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-livecodebench-v6-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",89.6,89.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:lcb-v6:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",90.3,90.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-livecodebench-v6","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",90.3,90.3,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:livecodebench:cell:language:lcb-v6:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",91.9,91.9,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["evidence-2026-08-15-solar-open-100b-reasoning-livecodebench-v6-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",56.5,56.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-livecodebench-v6","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team","v6",92.4,92.4,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-069","claude-haiku-4-5","Claude Haiku 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,51.11,51.11,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-claude-4-5-haiku-evals","aa-claude-4-5-haiku-evals","Artificial Analysis evaluations for claude-4-5-haiku","Artificial Analysis","https://artificialanalysis.ai/models/claude-4-5-haiku","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1405","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,63.5979,63.5979,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1395","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,65.3545,65.3545,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-578","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,84.8,84.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1777","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,47.3016,47.3016,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1728","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,65.5026,65.5026,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1382","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,71.4286,71.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1541","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,79.7884,79.7884,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1527","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,86.2434,86.2434,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-228","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,55.2,55.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-227","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,56.8,56.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1840","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,71.3228,71.3228,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1802","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,80.1058,80.1058,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-579","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,91.7,91.7,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1742","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,69.5238,69.5238,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1555","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,89.418,89.418,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1342","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,84.5503,84.5503,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1342--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,84.5503,84.5503,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1327","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,84.0212,84.0212,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1327--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,84.0212,84.0212,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1354","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,69.2063,69.2063,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1354--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,69.2063,69.2063,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1305","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,88.8889,88.8889,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1305--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,88.8889,88.8889,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1872","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,69.6296,69.6296,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1661","grok-4","Grok 4","Grok 4",null,"Grok 4","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,81.9048,81.9048,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1765","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,83.1746,83.1746,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1894","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,65.7143,65.7143,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1851","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,60.9524,60.9524,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1672","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,85.291,85.291,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-226","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,85,85,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1591","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,86.7725,86.7725,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1753","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,82.6455,82.6455,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1692","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,80.9524,80.9524,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-095","mistral-large-3","Mistral Large 3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,46.46,46.46,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-large-3-evals","aa-mistral-large-3-evals","Artificial Analysis evaluations for mistral-large-3","Artificial Analysis","https://artificialanalysis.ai/models/mistral-large-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1649","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,73.0159,73.0159,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1862","o1","o1","o1",null,"o1","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,67.9365,67.9365,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1367","o3","o3","o3",null,"o3","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,80.8466,80.8466,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1814","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,85.9259,85.9259,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1814--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,85.9259,85.9259,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-577","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,87.1,87.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-224","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,91.6,91.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-225","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","livecodebench","LiveCodeBench","coding","LiveCodeBench team",null,89.6,89.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["benchlm-ref-deepseek-v4-flash-high-livecodebenchpass1cot-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",88.4,86.6841,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-livecodebenchpass1cot-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",88.4,86.6841,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-livecodebenchpass1cot-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",91.6,95.0392,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-livecodebenchpass1cot-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",55.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-livecodebenchpass1cot-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",91.6,95.0392,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-livecodebenchpass1cot-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",89.8,90.3394,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-livecodebenchpass1cot-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",89.8,90.3394,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-livecodebenchpass1cot-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",93.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-livecodebenchpass1cot-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",56.8,4.1775,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-livecodebenchpass1cot-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpass1cot","LiveCodeBench Pass@1 with Chain-of-Thought","coding","DeepSeek","2026",93.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",70.7,70.4932,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",70.7,70.4932,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",70.7,70.4932,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",82.9,88.4028,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",82.9,88.4028,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",82.9,88.4028,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-livecodebenchpro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.5,95.1556,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-livecodebenchpro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.5,95.1556,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-livecodebenchpro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.5,95.1556,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",74.2,75.6312,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",74.2,75.6312,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",74.2,75.6312,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchpro-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",22.68,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchpro-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",22.68,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchpro-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",22.68,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-livecodebenchpro-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",80,84.1456,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-livecodebenchpro-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",80,84.1456,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-livecodebenchpro-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",80,84.1456,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchpro-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.8,95.596,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchpro-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.8,95.596,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchpro-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",87.8,95.596,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",90.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",90.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchpro","LiveCodeBench Pro","coding","LiveCodeBench Pro authors","2025",90.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-livecodebenchv5-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv5","LiveCodeBench v5","coding","LiveCodeBench maintainers","2025",63.2,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",84.8,85.9249,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",84.8,85.9249,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",84.8,85.9249,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-livecodebenchv6-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",72,64.4772,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-livecodebenchv6-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",85,86.2601,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-livecodebenchv6-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",85,86.2601,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-livecodebenchv6-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",85,86.2601,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-livecodebenchv6-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",89.6,93.9678,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-livecodebenchv6-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",89.6,93.9678,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-livecodebenchv6-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",89.6,93.9678,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-livecodebenchv6-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",87.7,90.7842,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-livecodebenchv6-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",37.2,6.1662,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-livecodebenchv6-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",69.9,60.9584,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchv6-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",33.52,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchv6-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",33.52,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-livecodebenchv6-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",33.52,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-livecodebenchv6-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",89,92.9625,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",83.6,83.9142,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",83.6,83.9142,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",83.6,83.9142,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",87.1,89.7788,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",87.1,89.7788,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",87.1,89.7788,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchv6-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",92.9,99.4973,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchv6-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",92.9,99.4973,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-livecodebenchv6-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",92.9,99.4973,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",93.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",93.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",93.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.7,53.9209,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.7,53.9209,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.7,53.9209,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-livecodebenchv6-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.8,54.0885,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-livecodebenchv6-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.8,54.0885,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-livecodebenchv6-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-livecodebenchv6","LiveCodeBench v6","coding","LiveCodeBench maintainers","2026",65.8,54.0885,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-meta-muse-12-meta-internal-coding-claude-opus-5","claude-opus-5","Claude Opus 5","Claude Opus 5 (reasoning configuration not stated in chart)","claude-opus-5-meta-muse-12-release-unspecified","Claude Opus 5 (reasoning configuration not stated in chart)","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",79.4,79.4,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-internal-coding-chart","refresh-meta-muse-spark-1-2-internal-coding-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/meta-internal-coding-bench-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-meta-internal-coding-gemini-3-6-flash","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (reasoning configuration not stated in chart)","gemini-3-6-flash-meta-muse-12-release-unspecified","Gemini 3.6 Flash (reasoning configuration not stated in chart)","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",63.9,63.9,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-internal-coding-chart","refresh-meta-muse-spark-1-2-internal-coding-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/meta-internal-coding-bench-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-meta-internal-coding-gpt-5-6-terra","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (reasoning configuration not stated in chart)","gpt-5-6-terra-meta-muse-12-release-unspecified","GPT-5.6 Terra (reasoning configuration not stated in chart)","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",65.4,65.4,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-internal-coding-chart","refresh-meta-muse-spark-1-2-internal-coding-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/meta-internal-coding-bench-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-meta-internal-coding-muse-spark-1-1","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (reasoning configuration not stated in chart)","muse-spark-1-1-meta-muse-12-release-unspecified","Muse Spark 1.1 (reasoning configuration not stated in chart)","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",68.3,68.3,"percent","higher","2.0.0","excluded","supported","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-internal-coding-chart","refresh-meta-muse-spark-1-2-internal-coding-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/meta-internal-coding-bench-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Complete comparison-table row retained for display and provenance; the chart does not state an exact reasoning configuration."],["evidence-2026-08-15-meta-muse-12-meta-internal-coding-muse-spark-1-2","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",70.6,70.6,"percent","higher","2.0.0","excluded","direct","2026-08-05","2026-08-05","production::refresh-meta-muse-spark-1-2-internal-coding-chart","refresh-meta-muse-spark-1-2-internal-coding-chart","Meta Superintelligence Labs permanent refresh source","Meta Superintelligence Labs","https://research.meta.ai/articles/introducing-muse-code-and-muse-spark-1-2/evaluations/meta-internal-coding-bench-v1.png","2026-08-05","2026-08-15","2026-08-15","provider-reported","Exact Muse Spark 1.2 result retained with the published xhigh setting."],["evidence-2026-08-muse-spark-1-2-meta-internal-coding","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) in Meta's internal harness",null,"Muse Spark 1.2 (xhigh) in Meta's internal harness","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",70.6,70.6,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 70.6% over two attempts on 440 internal codebase tasks. This result is displayed as provider-reported reference evidence only."],["evidence-2026-08-muse-spark-1-2-meta-internal-coding--configuration--muse-spark-1-2-xhigh","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh) in Meta's internal harness","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh) in Meta's internal harness","meta-internal-coding-bench","Meta Internal Coding Bench","coding","Meta","August 2026",70.6,70.6,"percent","higher","1.8.0","reference-only","direct","2026-08-05","2026-08-05","production::meta-muse-spark-1-2-methodology-2026-08-05","meta-muse-spark-1-2-methodology-2026-08-05","Muse Spark 1.2 Evaluation Methodology","Meta Superintelligence Labs","https://research.meta.ai/static/muse-spark-1-2-methodology","2026-08-05","2026-08-05","2026-08-05","provider-reported","Meta reports 70.6% over two attempts on 440 internal codebase tasks. This result is displayed as provider-reported reference evidence only."],["evidence-2026-07-2030","claude-sonnet-5","Claude Sonnet 5","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",66.9,66.9,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 MLE-Bench comparison score."],["evidence-2026-07-2012","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","As published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro comparison.",null,"As published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro comparison.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",42.6,42.6,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","MLE-Bench comparison row for 3.1 Pro."],["evidence-2026-07-2010","gemini-3-5-flash","Gemini 3.5 Flash","As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.",null,"As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",49.7,49.7,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","MLE-Bench comparison row for 3.5 Flash."],["evidence-2026-07-1991","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). MLE-Bench Partial-30 Average Position Score; k=2 independent runs; bash terminal harness.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). MLE-Bench Partial-30 Average Position Score; k=2 independent runs; bash terminal harness.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",63.9,63.9,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","MLE-Bench Average Position Score (Partial-30) from Google evaluation table."],["evidence-2026-07-2017","gpt-5-6-luna","GPT-5.6 Luna","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",47.6,47.6,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna MLE-Bench comparison score."],["evidence-2026-07-2024","grok-4-5","Grok 4.5","As published in Gemini 3.6 Flash evaluation table.",null,"As published in Gemini 3.6 Flash evaluation table.","mle-bench","MLE-Bench","coding","OpenAI","Partial-30",43.2,43.2,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Grok 4.5 MLE-Bench comparison score."],["benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","2026",35.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","2026",35.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","2026",35.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mlsbenchlite-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","2026",48.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mlsbenchlite-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","2026",48.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mlsbenchlite-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-mlsbenchlite","MLS-Bench Lite","coding","MLS-Bench","MLS-Bench-Lite",48.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports MLS-Bench-Lite=48.3 with the Kimi Code harness at max effort. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-minimax-m2-7-multiswebench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","multi-swe-bench","Multi-SWE-Bench","coding","Multi-SWE-Bench","2026",52.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-multiswebench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","multi-swe-bench","Multi-SWE-Bench","coding","Multi-SWE-Bench","2026",52.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-multiswebench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","multi-swe-bench","Multi-SWE-Bench","coding","Multi-SWE-Bench","2026",52.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-804","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","multi-swe-bench","Multi-SWE-Bench","coding","Multi-SWE-Bench",null,52.7,52.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-claude-opus-4-5-nl2repo-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",43.2,73.7327,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-nl2repo-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",43.2,73.7327,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-nl2repo-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",43.2,59.2593,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:nl2repo:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",47.6,47.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","NL2Repo-Bench uses Claude Code with pip install, direct downloads and git clone disabled. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-nl2repo-2026-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",47.6,47.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-opus-4-8-benchlm-nl2repo-2026-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",69.7,69.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["benchlm-ref-deepseek-v4-flash-max-nl2repo-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",54.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-nl2repo-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",54.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:nl2repo:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",54.2,54.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","NL2Repo-Bench uses Claude Code with pip install, direct downloads and git clone disabled. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["deepseek-v4-flash-vision-exp-nl2repo-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","deepseek-v4-flash-vision-exp-max-harness","DeepSeek Harness Minimal Mode; max effort; top_p=0.95; temperature=1.0","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",57.7,57.7,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value. Canonical NL2Repo identity, configuration, metric and unit are resolved; the current protocol registry retains this track as reference-only."],["evidence-2026-08-15-deepseek-v4-pro-0813-benchlm-nl2repo-2026-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",61.1,61.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["benchlm-ref-glm-5-1-nl2repo-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.7,71.4286,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-nl2repo-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.7,71.4286,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-nl2repo-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.7,57.4074,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-nl2repo-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-nl2repo-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-nl2repo-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.9,80.3704,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-glm-5-2-benchlm-nl2repo-2026-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.9,48.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["evidence-2026-08-15-glm-5-3-benchlm-nl2repo-2026","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",58,58,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["evidence-2026-08-15-kimi-k3-benchlm-nl2repo-2026-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",58,58,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["benchlm-ref-minimax-m2-7-nl2repo-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",39.8,58.0645,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-nl2repo-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",39.8,58.0645,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-nl2repo-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",39.8,46.6667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-nl2repo-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.13,68.8018,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-nl2repo-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.13,68.8018,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-nl2repo-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.13,55.2963,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-nl2repo-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",34.6,34.1014,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-nl2repo-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",34.6,34.1014,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-nl2repo-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",34.6,27.4074,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-nl2repo-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.2,96.7742,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-nl2repo-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.2,96.7742,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-nl2repo-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",48.2,77.7778,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-nl2repo-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",27.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-nl2repo-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",27.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-nl2repo-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",27.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-nl2repo-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.9,72.3502,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-nl2repo-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.9,72.3502,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-nl2repo-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.9,58.1481,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-nl2repo-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",36.2,41.4747,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-nl2repo-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",36.2,41.4747,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-nl2repo-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",36.2,33.3333,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-nl2repo-2026-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",36.2,36.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",29.4,10.1382,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",29.4,10.1382,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",29.4,8.1481,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nl2repo-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",47.2,92.1659,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nl2repo-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",47.2,92.1659,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nl2repo-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",47.2,74.0741,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nl2repo-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",41.1,64.0553,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nl2repo-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",41.1,64.0553,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nl2repo-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",41.1,51.4815,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:nl2repo:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",41.1,41.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","NL2Repo-Bench uses Claude Code with pip install, direct downloads and git clone disabled. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-nl2repo-2026-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",41.1,41.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-max-benchlm-nl2repo-2026-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",55.9,55.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: temp=1.0, top_p=1.0, max_new_tokens=64k, 1M context."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:nl2repo:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.3,42.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","NL2Repo-Bench uses Claude Code with pip install, direct downloads and git clone disabled. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-nl2repo-2026","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-nl2repo","NL2Repo","coding","MiniMax","2026",42.3,42.3,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["deepseek-v4-pro-0813-release-nl2repo-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",69.7,69.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-nl2repo-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",39.4,39.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-nl2repo-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",54.2,54.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-nl2repo-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",38.5,38.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-nl2repo-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",61.5,61.5,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-nl2repo-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-nl2repo","NL2Repo","coding","MiniMax","August 2026",48.9,48.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["benchlm-ref-kimi-3-posttrainbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","posttrain-bench","PostTrainBench","coding","PostTrainBench","2026",36.6,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-posttrainbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","posttrain-bench","PostTrainBench","coding","PostTrainBench","2026",36.6,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-posttrainbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","posttrain-bench","PostTrainBench","coding","PostTrainBench","2026",36.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-fable-5-posttrain-bench-public-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",41.8,41.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-opus-4-8-posttrain-bench-public-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",32.9,32.9,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-2-posttrain-bench-public-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",31.7,31.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-3-posttrain-bench-public","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",39.8,39.8,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-15-gpt-5-6-sol-posttrain-bench-public-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",36.2,36.2,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-1960","kimi-k3","Kimi K3","Claude Code harness; official Harbor implementation; max reasoning; average of three H20 runs.",null,"Claude Code harness; official Harbor implementation; max reasoning; average of three H20 runs.","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",36.6,36.6,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 36.6 on PostTrainBench."],["evidence-2026-07-1960--configuration--kimi-k3-max","kimi-k3","Kimi K3","Claude Code harness; official Harbor implementation; max reasoning; average of three H20 runs.","kimi-k3-max","Claude Code harness; official Harbor implementation; max reasoning; average of three H20 runs.","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",36.6,36.6,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 36.6 on PostTrainBench."],["evidence-2026-08-15-kimi-k3-posttrain-bench-public-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","posttrain-bench","PostTrainBench","coding","PostTrainBench","public",32,32,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-fable-5-program-bench-almost-solved-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",33,33,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-opus-4-8-program-bench-almost-solved-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",15.5,15.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-2-program-bench-almost-solved-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",9.5,9.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-3-program-bench-almost-solved","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",19,19,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-15-gpt-5-6-sol-program-bench-almost-solved-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",23,23,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-kimi-k3-program-bench-almost-solved-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",17.5,17.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-max-program-bench-almost-solved-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","program-bench","ProgramBench","coding","ProgramBench","Almost Solved",10.5,10.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-1957","kimi-k3","Kimi K3","KimiCode harness; ProgramBench raw hidden-test pass rate; max reasoning.",null,"KimiCode harness; ProgramBench raw hidden-test pass rate; max reasoning.","program-bench","ProgramBench","coding","ProgramBench","public",77.8,77.8,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Raw hidden-test pass rate 77.8, not fully-resolved task rate."],["evidence-2026-07-1957--configuration--kimi-k3-max","kimi-k3","Kimi K3","KimiCode harness; ProgramBench raw hidden-test pass rate; max reasoning.","kimi-k3-max","KimiCode harness; ProgramBench raw hidden-test pass rate; max reasoning.","program-bench","ProgramBench","coding","ProgramBench","public",77.8,77.8,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Raw hidden-test pass rate 77.8, not fully-resolved task rate."],["benchlm-ref-claude-opus-5-programbenchepisode1-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-programbenchepisode1","ProgramBench hidden-test pass rate after episode 1","coding","Yang et al.","2026",83,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-programbenchepisode1-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-programbenchepisode1","ProgramBench hidden-test pass rate after episode 1","coding","Yang et al.","2026",83,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-programbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",93,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-programbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",93,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-programbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",63.7,41.7355,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-programbench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",63.7,25.6345,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-programbench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",63.7,25.6345,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-programbench-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",53.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-programbench-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",53.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-programbench-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",53.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-programbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",77.8,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-programbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","2026",77.8,61.4213,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-programbench-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-programbench","ProgramBench: Can Language Models Rebuild Programs From Scratch?","coding","John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","ProgramBench public evaluation",77.8,61.4213,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports ProgramBench=77.8 with the Kimi Code harness at max effort. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["evidence-2026-08-15-claude-opus-4-6-qwen-swe-bench-2026-08-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","qwen-swe-bench","QwenSWEBench","coding","Qwen","2026-08",63.8,63.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","In-house QwenSWEBench; avg@3, 8-hour timeout, max_tokens=32768."],["evidence-2026-08-15-qwen3-6-27b-qwen-swe-bench-2026-08-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","qwen-swe-bench","QwenSWEBench","coding","Qwen","2026-08",49.3,49.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","In-house QwenSWEBench; avg@3, 8-hour timeout, max_tokens=32768."],["evidence-2026-08-15-qwen-3-7-plus-qwen-swe-bench-2026-08-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","qwen-swe-bench","QwenSWEBench","coding","Qwen","2026-08",59.2,59.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","In-house QwenSWEBench; avg@3, 8-hour timeout, max_tokens=32768."],["evidence-2026-08-15-qwen-3-8-27b-qwen-swe-bench-2026-08","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","qwen-swe-bench","QwenSWEBench","coding","Qwen","2026-08",79,79,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","In-house QwenSWEBench; avg@3, 8-hour timeout, max_tokens=32768."],["benchlm-ref-claude-opus-4-6-reactnativeevals-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.1,52.1912,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-reactnativeevals-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.1,52.1912,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-reactnativeevals-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.1,52.1912,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-reactnativeevals-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",82.8,47.012,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-reactnativeevals-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",82.8,47.012,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-reactnativeevals-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",82.8,47.012,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",80.6,38.247,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",80.6,38.247,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",80.6,38.247,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-reactnativeevals-2026-07-21","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",96.1,100,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-reactnativeevals-2026-07-27","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",96.1,100,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-reactnativeevals-2026-08-01","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",96.1,100,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-fast-reactnativeevals-2026-07-21","composer-2-fast","Composer 2 Fast","Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",94.9,95.2191,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-fast-reactnativeevals-2026-07-27","composer-2-fast","Composer 2 Fast","Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",94.9,95.2191,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-fast-reactnativeevals-2026-08-01","composer-2-fast","Composer 2 Fast","Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",94.9,95.2191,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-reactnativeevals-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.5,1.992,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-reactnativeevals-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.5,1.992,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-reactnativeevals-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.5,1.992,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",78.9,31.4741,"elo","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",78.9,31.4741,"elo","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",78.9,31.4741,"elo","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-reactnativeevals-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",75.2,16.7331,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-reactnativeevals-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",75.2,16.7331,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-reactnativeevals-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",75.2,16.7331,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reactnativeevals-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",74.8,15.1394,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reactnativeevals-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",74.8,15.1394,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reactnativeevals-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",74.8,15.1394,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-reactnativeevals-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",85.3,56.9721,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-reactnativeevals-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",85.3,56.9721,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-reactnativeevals-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",85.3,56.9721,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-reactnativeevals-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.7,54.5817,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-reactnativeevals-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.7,54.5817,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-reactnativeevals-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",84.7,54.5817,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-reactnativeevals-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.6,2.3904,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-reactnativeevals-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.6,2.3904,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-reactnativeevals-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.6,2.3904,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-reactnativeevals-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71,0,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-reactnativeevals-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71,0,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-reactnativeevals-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71,0,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-reactnativeevals-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",72.6,6.3745,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-reactnativeevals-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",72.6,6.3745,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-reactnativeevals-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",72.6,6.3745,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reactnativeevals-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",77.2,24.7012,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reactnativeevals-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",77.2,24.7012,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reactnativeevals-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",77.2,24.7012,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-reactnativeevals-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.4,1.5936,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-reactnativeevals-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.4,1.5936,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-reactnativeevals-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-reactnativeevals","React Native Evals","coding","Callstack","2026",71.4,1.5936,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["aa-current:claude-opus-4-6-thinking:scicode:2026-08-29","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",51.851851851852,51.851851851852,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-6-thinking-2026-08-29","aa-current-claude-opus-4-6-thinking-2026-08-29","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-6-adaptive","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:claude-opus-4-7-adaptive:scicode:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",54.513888888889,54.513888888889,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:deepseek-v3-1:scicode:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","deepseek-v3-1-non-reasoning","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",36.689814814815,36.689814814815,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-2026-08-29","aa-current-deepseek-v3-1-2026-08-29","DeepSeek V3.1 (Non-reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:deepseek-v3-1-reasoning:scicode:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","deepseek-v3-1-reasoning-default","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",39.12037037037,39.12037037037,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-reasoning-2026-08-29","aa-current-deepseek-v3-1-reasoning-2026-08-29","DeepSeek V3.1 (Reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-individual:deepseek-v4-flash-vision-exp:scicode:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",46.6435185185185,46.6435185185185,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual scicode result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gemma-3-27b:scicode:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",21.180555555556,21.180555555556,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:gemma-4-26b-a4b:scicode:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",40.046296296296,40.046296296296,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:glm-5-2:scicode:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",50.462962962963,50.462962962963,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-individual:glm-5-3-flash:scicode:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",46.0648148148148,46.0648148148148,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual scicode result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:scicode:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",40.393518518518,40.393518518518,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:gpt-4-1-nano:scicode:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",25.925925925926,25.925925925926,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:kimi-k2-5-reasoning:scicode:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",48.958333333333,48.958333333333,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:llama-4-maverick:scicode:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",33.101851851852,33.101851851852,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:llama-4-scout:scicode:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",17.013888888889,17.013888888889,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:longcat-2-0:scicode:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",35.416666666667,35.416666666667,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:mistral-medium-3-5-128b:scicode:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",39.583333333333,39.583333333333,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:mistral-small-4-reasoning:scicode:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",37.962962962963,37.962962962963,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-individual:muse-spark-1-2:scicode:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",56.3657407407407,56.3657407407407,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual scicode result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:scicode:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",29.62962962963,29.62962962963,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:qwen3-5-397b-reasoning:scicode:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",42.013888888889,42.013888888889,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:qwen3-5-122b-a10b:scicode:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",42.013888888889,42.013888888889,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:qwen3-6-35b-a3b:scicode:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",35.763888888889,35.763888888889,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:qwen-3-8-flash-next:scicode:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","2024",46.875,46.875,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["aa-current:trinity-large-thinking:scicode:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","scicode","SciCode","coding","SciCode authors","2024",36.111111111111,36.111111111111,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","SciCode remains reference-only under the frozen LuminaBench methodology."],["benchlm-ref-claude-3-haiku-aascicode-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",18.6,29.8482,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aascicode-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",18.6,29.8482,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aascicode-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",18.6,29.8482,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aascicode-2026-07-21","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",23.3,37.774,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aascicode-2026-07-27","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",23.3,37.774,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aascicode-2026-08-01","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",23.3,37.774,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aascicode-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.3,61.3828,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aascicode-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.3,61.3828,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aascicode-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.3,61.3828,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aascicode-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",60.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aascicode-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",60.2,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aascicode-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",60.2,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aascicode-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aascicode-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aascicode-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.5,81.9562,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.5,81.9562,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.5,81.9562,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aascicode-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.9,86.0034,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aascicode-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.9,86.0034,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aascicode-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.9,86.0034,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aascicode-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aascicode-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aascicode-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aascicode-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.5,90.3879,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aascicode-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.5,90.3879,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aascicode-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.5,90.3879,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aascicode-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.1,82.968,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aascicode-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.1,82.968,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aascicode-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.1,82.968,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aascicode-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aascicode-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aascicode-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aascicode-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",55.7,92.4115,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aascicode-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",55.7,92.4115,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aascicode-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aascicode-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aascicode-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aascicode-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.6,88.8702,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aascicode-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.6,88.8702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aascicode-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.6,88.8702,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aascicode-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.8,62.226,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aascicode-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.8,62.226,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aascicode-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.8,62.226,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-07-21","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.6,61.8887,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-07-27","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.6,61.8887,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-08-01","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.6,61.8887,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aascicode-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.4,58.1788,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aascicode-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.4,58.1788,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aascicode-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.4,58.1788,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aascicode-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.1,64.4182,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aascicode-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.1,64.4182,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aascicode-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.1,64.4182,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aascicode-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aascicode-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aascicode-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aascicode-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.7,63.7437,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aascicode-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.7,63.7437,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aascicode-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.7,63.7437,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aascicode-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aascicode-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.9,74.199,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aascicode-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.4,76.7285,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aascicode-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50,82.7993,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aascicode-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.3,66.4418,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aascicode-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.3,66.4418,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aascicode-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.3,66.4418,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aascicode-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.4,10.9612,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aascicode-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.4,10.9612,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aascicode-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.4,10.9612,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aascicode-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.2,40.9781,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aascicode-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.2,40.9781,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aascicode-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.2,40.9781,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aascicode-2026-07-21","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",11.7,18.2125,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aascicode-2026-07-27","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",11.7,18.2125,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aascicode-2026-08-01","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",11.7,18.2125,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aascicode-2026-07-21","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.5,48.2293,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aascicode-2026-07-27","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.5,48.2293,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aascicode-2026-08-01","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.5,48.2293,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aascicode-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.1,47.5548,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aascicode-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.1,47.5548,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aascicode-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.1,47.5548,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aascicode-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.8,70.6577,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aascicode-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.8,70.6577,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aascicode-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.8,70.6577,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aascicode-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aascicode-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aascicode-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aascicode-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aascicode-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aascicode-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.9,69.14,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.9,69.14,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.9,69.14,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aascicode-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.9,97.8078,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aascicode-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.9,97.8078,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aascicode-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.9,97.8078,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aascicode-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.1,88.027,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aascicode-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.1,88.027,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aascicode-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.1,88.027,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.9,67.4536,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aascicode-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.7,87.3524,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aascicode-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.7,87.3524,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aascicode-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.7,87.3524,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aascicode-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",21.2,34.2327,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aascicode-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",21.2,34.2327,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aascicode-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",21.2,34.2327,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aascicode-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.2,62.9005,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aascicode-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.2,62.9005,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aascicode-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.2,62.9005,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aascicode-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aascicode-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aascicode-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aascicode-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.4,71.6695,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aascicode-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.4,71.6695,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aascicode-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.4,71.6695,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aascicode-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.9,33.7268,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aascicode-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.9,33.7268,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aascicode-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.9,33.7268,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aascicode-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.4,39.629,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aascicode-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.4,39.629,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aascicode-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.4,39.629,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aascicode-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",30.6,50.0843,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aascicode-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",30.6,50.0843,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aascicode-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",30.6,50.0843,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aascicode-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aascicode-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aascicode-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aascicode-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.1,74.5363,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aascicode-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.1,74.5363,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aascicode-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.1,74.5363,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aascicode-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.2,76.3912,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aascicode-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.2,76.3912,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aascicode-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.2,76.3912,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aascicode-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.6,72.0067,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aascicode-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.6,72.0067,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aascicode-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.6,72.0067,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aascicode-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.8,72.344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aascicode-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.8,72.344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aascicode-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.8,72.344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aascicode-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.5,83.6425,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aascicode-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.5,83.6425,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aascicode-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.5,83.6425,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aascicode-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.5,71.8381,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aascicode-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.5,71.8381,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aascicode-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.5,71.8381,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aascicode-2026-07-21","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",31.9,52.2766,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aascicode-2026-07-27","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",31.9,52.2766,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aascicode-2026-08-01","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",31.9,52.2766,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aascicode-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.1,62.7319,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aascicode-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.1,62.7319,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aascicode-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.1,62.7319,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aascicode-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.4,66.6105,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aascicode-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.4,66.6105,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aascicode-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.4,66.6105,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aascicode-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aascicode-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aascicode-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aascicode-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.3,54.6374,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aascicode-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.3,54.6374,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aascicode-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.3,54.6374,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aascicode-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",22.9,37.0995,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aascicode-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",22.9,37.0995,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aascicode-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",22.9,37.0995,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aascicode-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.9,70.8263,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aascicode-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41.1,67.7909,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aascicode-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.3,71.5008,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aascicode-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.3,71.5008,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aascicode-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",43.3,71.5008,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aascicode-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aascicode-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aascicode-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aascicode-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aascicode-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aascicode-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.2,66.2732,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aascicode-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.1,86.3406,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aascicode-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.1,86.3406,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aascicode-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.1,86.3406,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aascicode-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.6,90.5565,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aascicode-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.6,90.5565,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aascicode-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.6,90.5565,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aascicode-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.2,88.1956,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aascicode-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.2,88.1956,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aascicode-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.2,88.1956,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aascicode-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.6,93.9292,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aascicode-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.6,93.9292,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aascicode-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.6,93.9292,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aascicode-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aascicode-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aascicode-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49.9,82.6307,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aascicode-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aascicode-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aascicode-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aascicode-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aascicode-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aascicode-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aascicode-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.5,87.0152,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aascicode-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.5,87.0152,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aascicode-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",52.5,87.0152,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aascicode-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aascicode-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aascicode-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",56.1,93.086,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aascicode-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.9,89.3761,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aascicode-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.9,89.3761,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aascicode-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.9,89.3761,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aascicode-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.9,64.0809,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aascicode-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.9,64.0809,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aascicode-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.9,64.0809,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aascicode-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.4,56.4924,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aascicode-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.4,56.4924,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aascicode-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.4,56.4924,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aascicode-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.7,13.1535,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aascicode-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.7,13.1535,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aascicode-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.7,13.1535,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aascicode-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",0.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aascicode-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",0.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aascicode-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",0.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aascicode-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.2,12.3103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aascicode-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.2,12.3103,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aascicode-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",8.2,12.3103,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aascicode-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",1.7,1.3491,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aascicode-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",1.7,1.3491,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aascicode-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",1.7,1.3491,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aascicode-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aascicode-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aascicode-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.7,75.5481,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aascicode-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aascicode-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aascicode-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aascicode-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aascicode-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aascicode-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",44.2,73.0185,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aascicode-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.3,78.2462,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aascicode-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.3,78.2462,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aascicode-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.3,78.2462,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aascicode-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.1,89.7133,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aascicode-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.1,89.7133,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aascicode-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",54.1,89.7133,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aascicode-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aascicode-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aascicode-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aascicode-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aascicode-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aascicode-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aascicode-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aascicode-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aascicode-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.6,78.7521,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aascicode-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.1,76.2226,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aascicode-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.1,76.2226,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aascicode-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.1,76.2226,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aascicode-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.6,58.516,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aascicode-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.6,58.516,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aascicode-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.6,58.516,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aascicode-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.5,56.661,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aascicode-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.5,56.661,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aascicode-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.5,56.661,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aascicode-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aascicode-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aascicode-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aascicode-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aascicode-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aascicode-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",49,81.113,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aascicode-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aascicode-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aascicode-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",53.5,88.7015,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aascicode-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.5,78.5835,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aascicode-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.5,78.5835,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aascicode-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47.5,78.5835,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aascicode-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.7,97.4705,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aascicode-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.7,97.4705,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.8,11.6358,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.8,11.6358,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",7.8,11.6358,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",3,3.5413,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",3,3.5413,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",3,3.5413,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aascicode-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aascicode-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aascicode-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aascicode-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.9,48.9039,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aascicode-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.9,48.9039,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aascicode-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.9,48.9039,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aascicode-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aascicode-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aascicode-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aascicode-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",17,27.1501,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aascicode-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",17,27.1501,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aascicode-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",17,27.1501,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aascicode-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aascicode-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aascicode-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",25.9,42.1585,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aascicode-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aascicode-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aascicode-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.7,60.371,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aascicode-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.5,70.1518,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aascicode-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.5,70.1518,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aascicode-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42.5,70.1518,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aascicode-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.2,83.1366,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aascicode-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.2,83.1366,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aascicode-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",50.2,83.1366,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aascicode-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aascicode-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aascicode-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",47,77.7403,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aascicode-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.4,75.0422,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aascicode-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.4,75.0422,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aascicode-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.4,75.0422,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aascicode-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.2,47.7234,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aascicode-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.2,47.7234,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aascicode-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.2,47.7234,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aascicode-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aascicode-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aascicode-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.2,59.5278,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aascicode-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aascicode-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aascicode-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",33.1,54.3002,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.6,65.2614,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.6,65.2614,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.6,65.2614,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aascicode-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aascicode-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aascicode-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aascicode-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aascicode-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aascicode-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38,62.5632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aascicode-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.5,85.3288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aascicode-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.5,85.3288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aascicode-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",51.5,85.3288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aascicode-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.2,96.6273,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aascicode-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.2,96.6273,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aascicode-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",58.2,96.6273,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aascicode-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aascicode-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aascicode-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",29.6,48.398,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.8,45.3626,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.8,45.3626,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.8,45.3626,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aascicode-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aascicode-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aascicode-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aascicode-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.7,56.9983,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aascicode-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.7,56.9983,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aascicode-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",34.7,56.9983,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aascicode-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.8,33.5582,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aascicode-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.8,33.5582,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aascicode-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",20.8,33.5582,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aascicode-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aascicode-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aascicode-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aascicode-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41,67.6223,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aascicode-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41,67.6223,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aascicode-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",41,67.6223,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aascicode-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aascicode-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aascicode-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.9,65.7673,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aascicode-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26,42.3272,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aascicode-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26,42.3272,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aascicode-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26,42.3272,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aascicode-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aascicode-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aascicode-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",46.9,77.5717,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-07-21","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-07-27","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-08-01","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",27.1,44.1821,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aascicode-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.3,63.0691,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aascicode-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.3,63.0691,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aascicode-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",38.3,63.0691,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aascicode-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.5,65.0927,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aascicode-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.5,65.0927,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aascicode-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.5,65.0927,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aascicode-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aascicode-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aascicode-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aascicode-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aascicode-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aascicode-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",42,69.3086,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.7,62.0573,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.7,62.0573,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",37.7,62.0573,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aascicode-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.8,65.5987,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aascicode-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.8,65.5987,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aascicode-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",39.8,65.5987,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aascicode-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.7,67.1164,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aascicode-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.7,67.1164,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aascicode-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40.7,67.1164,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",35.8,58.8533,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aascicode-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",48.8,80.7757,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aascicode-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",48.8,80.7757,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aascicode-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",48.8,80.7757,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aascicode-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.5,75.2108,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aascicode-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.5,75.2108,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aascicode-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",45.5,75.2108,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aascicode-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26.4,43.0017,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aascicode-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26.4,43.0017,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aascicode-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",26.4,43.0017,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aascicode-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",19.2,30.86,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aascicode-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",19.2,30.86,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aascicode-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",19.2,30.86,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aascicode-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.8,40.3035,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aascicode-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.8,40.3035,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aascicode-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",24.8,40.3035,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aascicode-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aascicode-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aascicode-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",40,65.9359,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aascicode-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aascicode-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aascicode-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aascicode-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aascicode-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aascicode-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","scicode","SciCode","coding","SciCode authors","2026",36.1,59.3592,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-scicode-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","scicode","SciCode","coding","SciCode authors","288 test subproblems",43.6,43.6,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","No-tool, subproblem-level score; existing registry classifies SciCode as not weighted."],["evidence-2026-08-muse-glimmer-30b-scicode-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","scicode","SciCode","coding","SciCode authors","288 test subproblems",43.6,43.6,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","No-tool, subproblem-level score; existing registry classifies SciCode as not weighted."],["benchlm-ref-kimi-3-aascicode-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","scicode","SciCode","coding","SciCode authors","Artificial Analysis current as of 2026-07-23",58.7,97.4705,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports SciCode=58.7, cites Artificial Analysis as of 2026-07-23, and applies the card's all-results max-effort rule. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["aa-individual:claude-fable-5:scicode:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",60.1851851851852,60.1851851851852,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-4-8:scicode:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",53.4722222222222,53.4722222222222,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:scicode:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",55.671296296296305,55.671296296296305,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:claude-sonnet-5:scicode:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",53.587962962963,53.587962962963,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:deepseek-v4-pro-0813:scicode:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",49.189814814814795,49.189814814814795,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-6-flash:scicode:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",52.662037037037,52.662037037037,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:scicode:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",57.8703703703704,57.8703703703704,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3:scicode:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",56.4814814814815,56.4814814814815,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-4:scicode:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",56.5972222222222,56.5972222222222,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-5:scicode:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",56.1342592592593,56.1342592592593,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-luna:scicode:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",52.546296296296305,52.546296296296305,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-sol:scicode:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",56.1342592592593,56.1342592592593,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-terra:scicode:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",53.9351851851852,53.9351851851852,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-5:scicode:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",54.050925925925895,54.050925925925895,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-6:scicode:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",53.587962962963,53.587962962963,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:kimi-k3:scicode:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",58.6805555555556,58.6805555555556,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["aa-individual:muse-spark-1-1:scicode:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","scicode","SciCode","coding","SciCode authors","rolling",58.2175925925926,58.2175925925926,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. scicode is a non-ranking/reference benchmark under frozen Methodology and remains public reference evidence. The composite Intelligence Index is not ingested."],["evidence-2026-07-312","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,60.2,60.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1402","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","scicode","SciCode","coding","SciCode authors",null,39.8148,39.8148,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1391","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.8979,40.8979,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-681","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,47,47,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-686","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,45.7,45.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-679","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,50.1,50.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-320","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.5,53.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1774","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.2778,40.2778,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1724","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.0463,40.0463,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1378","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","scicode","SciCode","coding","SciCode authors",null,44.6759,44.6759,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-683","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,46.9,46.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-319","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.6,53.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1632","command-a-plus","Command A+","Command A+",null,"Command A+","scicode","SciCode","coding","SciCode authors",null,37.8472,37.8472,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1538","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.625,40.625,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1524","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","scicode","SciCode","coding","SciCode authors",null,38.8889,38.8889,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1836","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.5093,40.5093,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1798","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","scicode","SciCode","coding","SciCode authors",null,42.8241,42.8241,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-680","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,49.9,49.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-675","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,56.1,56.1,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-324","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,41.9,41.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-313","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,58.9,58.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-265","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.1,53.1,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2042","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis SciCode independent evaluation.",null,"Artificial Analysis SciCode independent evaluation.","scicode","SciCode","coding","SciCode authors",null,40.9,40.9,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA SciCode 40.9%."],["evidence-2026-07-2002","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis SciCode independent evaluation.",null,"Artificial Analysis SciCode independent evaluation.","scicode","SciCode","coding","SciCode authors",null,52.7,52.7,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA SciCode 52.7%."],["evidence-2026-07-1880","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,38.1944,38.1944,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1619","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,40.0463,40.0463,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-690","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,43.4,43.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1739","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","scicode","SciCode","coding","SciCode authors",null,38.4259,38.4259,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1552","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","scicode","SciCode","coding","SciCode authors",null,45.1389,45.1389,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-685","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,46.2,46.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-688","glm-5-turbo","GLM-5-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,43.6,43.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-687","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,43.8,43.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-677","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,50.5,50.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-689","glm-5v-turbo","GLM-5V-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,43.5,43.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1338","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","scicode","SciCode","coding","SciCode authors",null,42.9398,42.9398,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1338--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","scicode","SciCode","coding","SciCode authors",null,42.9398,42.9398,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1323","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","scicode","SciCode","coding","SciCode authors",null,40.8565,40.8565,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1323--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","scicode","SciCode","coding","SciCode authors",null,40.8565,40.8565,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1350","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","scicode","SciCode","coding","SciCode authors",null,40.9722,40.9722,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1350--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","scicode","SciCode","coding","SciCode authors",null,40.9722,40.9722,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-691","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,43.3,43.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1302","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","scicode","SciCode","coding","SciCode authors",null,52.0833,52.0833,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1302--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","scicode","SciCode","coding","SciCode authors",null,52.0833,52.0833,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1313","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","scicode","SciCode","coding","SciCode authors",null,54.6296,54.6296,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1313--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","scicode","SciCode","coding","SciCode authors",null,54.6296,54.6296,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-321","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.2,53.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-314","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,56.6,56.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1292","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","scicode","SciCode","coding","SciCode authors",null,49.8843,49.8843,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1292--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","scicode","SciCode","coding","SciCode authors",null,49.8843,49.8843,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-684","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,46.9,46.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-315","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,56.1,56.1,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-322","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,52.5,52.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-316","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,56.1,56.1,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-318","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.9,53.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1869","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","scicode","SciCode","coding","SciCode authors",null,40.625,40.625,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1657","grok-4","Grok 4","Grok 4",null,"Grok 4","scicode","SciCode","coding","SciCode authors",null,45.7176,45.7176,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1761","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","scicode","SciCode","coding","SciCode authors",null,44.213,44.213,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1446","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","scicode","SciCode","coding","SciCode authors",null,45.6019,45.6019,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-268","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,47.3,47.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-317","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,54.1,54.1,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1891","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","scicode","SciCode","coding","SciCode authors",null,36.2269,36.2269,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1848","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","scicode","SciCode","coding","SciCode authors",null,30.6713,30.6713,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1669","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","scicode","SciCode","coding","SciCode authors",null,42.3611,42.3611,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-267","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,48.7,48.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1420","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","scicode","SciCode","coding","SciCode authors",null,53.4722,53.4722,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1436","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","scicode","SciCode","coding","SciCode authors",null,47.4537,47.4537,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1946","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis SciCode evaluation.",null,"Kimi K3; Artificial Analysis SciCode evaluation.","scicode","SciCode","coding","SciCode authors",null,58.7,58.7,"percent","higher","1.4.1","reference-only","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis SciCode score mirrored via BenchLM public row for Kimi K3."],["evidence-2026-07-1786","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","scicode","SciCode","coding","SciCode authors",null,37.037,37.037,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1588","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","scicode","SciCode","coding","SciCode authors",null,39.3519,39.3519,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-694","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,36.7,36.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-692","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,42.5,42.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1575","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","scicode","SciCode","coding","SciCode authors",null,43.0556,43.0556,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-678","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,50.2,50.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1750","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","scicode","SciCode","coding","SciCode authors",null,36.1111,36.1111,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1689","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","scicode","SciCode","coding","SciCode authors",null,40.7407,40.7407,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1563","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","scicode","SciCode","coding","SciCode authors",null,42.5926,42.5926,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-682","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,47,47,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-323","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,45.4,45.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-326","mistral-large-3","Mistral Large 3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,36.2,36.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-325","mistral-small-4","Mistral Small 4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,38,38,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-676","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,51.5,51.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-674","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,58.2,58.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1914","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","scicode","SciCode","coding","SciCode authors",null,58.2,58.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1914--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","scicode","SciCode","coding","SciCode authors",null,58.2,58.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1827","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,35.9954,35.9954,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1605","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,39.9306,39.9306,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1645","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","scicode","SciCode","coding","SciCode authors",null,42.7083,42.7083,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-scicode-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","scicode","SciCode","coding","SciCode authors",null,32.6,32.6,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1859","o1","o1","o1",null,"o1","scicode","SciCode","coding","SciCode authors",null,35.7639,35.7639,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1363","o3","o3","o3",null,"o3","scicode","SciCode","coding","SciCode authors",null,40.9722,40.9722,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1810","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","scicode","SciCode","coding","SciCode authors",null,46.5278,46.5278,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1810--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","scicode","SciCode","coding","SciCode authors",null,46.5278,46.5278,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1680","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","scicode","SciCode","coding","SciCode authors",null,43.0556,43.0556,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1499","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,42.0139,42.0139,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1510","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,39.4676,39.4676,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1711","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,37.7315,37.7315,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1485","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,42.0139,42.0139,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1700","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","scicode","SciCode","coding","SciCode authors",null,40.5093,40.5093,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1459","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","scicode","SciCode","coding","SciCode authors",null,39.8148,39.8148,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1469","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","scicode","SciCode","coding","SciCode authors",null,46.875,46.875,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-693","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,40.7,40.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-264","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,53.5,53.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-266","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","scicode","SciCode","coding","SciCode authors",null,51.3,51.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["benchlm-ref-gemini-3-5-flash-scicode-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.1,78.852,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-scicode-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.1,78.852,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-scicode-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.1,78.852,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-scicode-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47.3,61.3293,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-scicode-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47.3,61.3293,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-scicode-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47.3,61.3293,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-scicode-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",41.2,42.9003,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-scicode-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",41.2,42.9003,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-scicode-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",41.2,42.9003,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-scicode-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",48.7,65.5589,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-scicode-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",48.7,65.5589,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-scicode-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",48.7,65.5589,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-scicode-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",48.7,65.5589,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-scicode-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",52.2,76.1329,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-scicode-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",52.2,76.1329,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-scicode-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",52.2,76.1329,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-scicode-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",27,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-scicode-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",27,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-scicode-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",27,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",32,15.1057,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",32,15.1057,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",32,15.1057,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-scicode-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",44.6,53.1722,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-scicode-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",44.6,53.1722,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-scicode-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",44.6,53.1722,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-scicode-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47,60.423,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-scicode-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47,60.423,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-scicode-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",47,60.423,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-scicode-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.5,80.0604,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-scicode-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.5,80.0604,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-scicode-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",53.5,80.0604,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-scicode-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",51.3,73.4139,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-scicode-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",51.3,73.4139,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-scicode-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",51.3,73.4139,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-scicode-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",60.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-scicode-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",60.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-scicode-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",60.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-scicode-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",58.7,95.7704,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-scicode-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",58.7,95.7704,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-scicode-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-scicode","Scientific Code Benchmark","coding","BenchLM registry","2024",58.7,95.7704,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-spider2lite-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-spider2lite","Spider 2.0-Lite","coding","Spider 2.0 authors","2024",52.9,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-spider2lite-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-spider2lite","Spider 2.0-Lite","coding","Spider 2.0 authors","2024",52.9,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-spider2lite-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-spider2lite","Spider 2.0-Lite","coding","Spider 2.0 authors","2024",52.9,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-svgbench-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-svgbench","SVG-Bench","coding","MiniMax","2026",63.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-svgbench-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-svgbench","SVG-Bench","coding","MiniMax","2026",63.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-svgbench-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-svgbench","SVG-Bench","coding","MiniMax","2026",63.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-fable-5-swe-marathon-1-1-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",33.1,33.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-claude-opus-4-8-swe-marathon-1-1-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",48.8,48.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-2-swe-marathon-1-1-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",19.4,19.4,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-glm-5-3-swe-marathon-1-1","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",42.5,42.5,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-15-gpt-5-6-sol-swe-marathon-1-1-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",42.5,42.5,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-kimi-k3-swe-marathon-1-1-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","swe-marathon","SWE Marathon","coding","Abundant AI","1.1",48.1,48.1,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-kimi-3-swemarathon-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","swe-marathon","SWE Marathon","coding","Abundant AI","2026",42,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-swemarathon-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","swe-marathon","SWE Marathon","coding","Abundant AI","2026",42,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-swemarathon-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","swe-marathon","SWE Marathon","coding","Abundant AI","2026",42,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1958","kimi-k3","Kimi K3","Claude Code harness; SWE Marathon v1.1 H20-calibrated branch; max reasoning.",null,"Claude Code harness; SWE Marathon v1.1 H20-calibrated branch; max reasoning.","swe-marathon","SWE Marathon","coding","Abundant AI","v1.1",42,42,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Provider-run evaluation; official public board listing pending at check date."],["evidence-2026-07-1958--configuration--kimi-k3-max","kimi-k3","Kimi K3","Claude Code harness; SWE Marathon v1.1 H20-calibrated branch; max reasoning.","kimi-k3-max","Claude Code harness; SWE Marathon v1.1 H20-calibrated branch; max reasoning.","swe-marathon","SWE Marathon","coding","Abundant AI","v1.1",42,42,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Provider-run evaluation; official public board listing pending at check date."],["benchlm-ref-claude-mythos-5-swemultimodal-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",54.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swemultimodal-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",54.9,85.623,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swemultimodal-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",54.9,85.623,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swemultimodal-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",38.4,38.4328,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swemultimodal-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",38.4,32.9073,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swemultimodal-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",38.4,32.9073,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swemultimodal-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",59.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swemultimodal-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",59.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultimodal-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",28.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultimodal-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",28.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultimodal-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","2025",28.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-claude-opus-4-6-benchlm-swemultimodal-swe-mm-public-dev-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","SWE-MM public dev",27.1,27.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-swemultimodal-swe-mm-public-dev-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","SWE-MM public dev",25.7,25.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-swemultimodal-swe-mm-public-dev-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","SWE-MM public dev",30,30,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-swemultimodal-swe-mm-public-dev","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-swemultimodal","SWE-bench Multimodal","coding","SWE-bench team","SWE-MM public dev",38.6,38.6,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-fable-swepro-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80,99.1979,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-swepro-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80,99.1979,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-swepro-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80,99.1979,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swepro-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swepro-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swepro-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",80.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swepro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.1,37.9679,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swepro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.1,37.9679,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swepro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.1,37.9679,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swepro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.4,28.0749,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swepro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.4,28.0749,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swepro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.4,28.0749,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-swepro-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.3,57.2193,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-swepro-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.3,57.2193,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-swepro-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.3,57.2193,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swepro-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",69.2,70.3209,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swepro-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",69.2,70.3209,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swepro-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",69.2,70.3209,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swepro-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",79.2,97.0588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swepro-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",79.2,97.0588,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swepro-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.2,54.2781,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swepro-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.2,54.2781,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swepro-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.2,54.2781,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swepro-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.3,25.1337,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swepro-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.1,16.5775,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swepro-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.1,16.5775,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swepro-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.1,16.5775,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swepro-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.6,25.9358,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swepro-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.4,30.7487,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swepro-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.1,24.5989,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swepro-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.1,24.5989,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swepro-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.1,24.5989,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swepro-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.4,33.4225,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-swepro-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-swepro-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-swepro-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-swepro-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.2,30.2139,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-swepro-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.2,30.2139,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-swepro-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.2,30.2139,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swepro-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swepro-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swepro-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.1,32.6203,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swepro-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.4,41.4439,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swepro-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.4,41.4439,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swepro-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.4,41.4439,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-swepro-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.1,51.3369,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-swepro-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.1,51.3369,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-swepro-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.1,51.3369,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-swepro-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.6,33.9572,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-swepro-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.6,33.9572,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-swepro-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.6,33.9572,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swepro-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.8,37.1658,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swepro-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.8,37.1658,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swepro-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.8,37.1658,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-swepro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.7,39.5722,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-swepro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.7,39.5722,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-swepro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.7,39.5722,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-swepro-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-swepro-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-swepro-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-swepro-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.7,52.9412,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-swepro-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.7,52.9412,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-swepro-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.7,52.9412,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-swepro-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.6,58.0214,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-swepro-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.6,58.0214,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-swepro-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.6,58.0214,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-swepro-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.4,54.8128,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-swepro-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.4,54.8128,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-swepro-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",63.4,54.8128,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-swepro-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",51.8,23.7968,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-swepro-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",51.8,23.7968,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-swepro-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",51.8,23.7968,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swepro-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.7,58.2888,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swepro-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.7,58.2888,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swepro-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",64.7,58.2888,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-swepro-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.3,30.4813,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-swepro-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.3,30.4813,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-swepro-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",54.3,30.4813,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-swepro-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",55.9,34.7594,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swepro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.7,20.8556,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swepro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.7,20.8556,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swepro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.7,20.8556,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swepro-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swepro-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swepro-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",58.6,41.9786,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swepro-2026-07-21","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.2,16.8449,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swepro-2026-07-27","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.2,16.8449,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swepro-2026-08-01","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.2,16.8449,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swepro-2026-07-21","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59.4,44.1176,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swepro-2026-07-27","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59.4,44.1176,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swepro-2026-08-01","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59.4,44.1176,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-swepro-2026-07-21","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",46.3,9.0909,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-swepro-2026-07-27","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",46.3,9.0909,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-swepro-2026-08-01","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",46.3,9.0909,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-swepro-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.8,26.4706,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-swepro-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.8,26.4706,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-swepro-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.8,26.4706,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-swepro-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.1,35.2941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-swepro-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.1,35.2941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-swepro-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.1,35.2941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-swepro-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.2,38.2353,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-swepro-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.2,38.2353,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-swepro-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.2,38.2353,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swepro-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.2,35.5615,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swepro-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.2,35.5615,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swepro-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.2,35.5615,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-swepro-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-swepro-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-swepro-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-swepro-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.4,25.4011,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-swepro-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.4,25.4011,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-swepro-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",52.4,25.4011,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-swepro-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",61.5,49.7326,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-swepro-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",61.5,49.7326,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-swepro-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",61.5,49.7326,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-swepro-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.4,20.0535,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-swepro-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.4,20.0535,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-swepro-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.4,20.0535,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swepro-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.2,51.6043,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swepro-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.2,51.6043,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swepro-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",62.2,51.6043,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swepro-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",42.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swepro-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",42.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swepro-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",42.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-swepro-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.3,38.5027,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-swepro-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.3,38.5027,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-swepro-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.3,38.5027,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-swepro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.9,21.3904,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-swepro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.9,21.3904,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-swepro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",50.9,21.3904,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swepro-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.5,28.3422,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swepro-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.5,28.3422,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swepro-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",53.5,28.3422,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swepro-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.6,36.631,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swepro-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.6,36.631,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swepro-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.6,36.631,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swepro-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.5,17.6471,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swepro-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.5,17.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swepro-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",49.5,17.6471,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swepro-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",60.6,47.3262,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swepro-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",60.6,47.3262,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swepro-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",60.6,47.3262,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swepro-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.6,39.3048,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swepro-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.6,39.3048,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swepro-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",57.6,39.3048,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-swepro-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-swepro-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-swepro-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",59,43.0481,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-swepro-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",73.7,82.3529,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-swepro-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",73.7,82.3529,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-swepro-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",73.7,82.3529,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-swepro-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.3,35.8289,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-swepro-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.3,35.8289,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-swepro-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","2025",56.3,35.8289,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-laguna-xs-2-1-swe-bench-pro-public-dataset","laguna-xs-2-1","Laguna XS 2.1","Laguna XS 2.1 (thinking enabled)","laguna-xs-2-1-default","Laguna XS 2.1 (thinking enabled)","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","Public Dataset",47.6,47.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-07-02","2026-07-02","production::poolside-laguna-xs-2-1-huggingface-2026-08-15","poolside-laguna-xs-2-1-huggingface-2026-08-15","Laguna XS 2.1 model card","Poolside","https://huggingface.co/poolside/Laguna-XS-2.1","2026-07-02","2026-08-15","2026-08-15","provider-reported","Official owner score; mean pass@1 over 2 attempts on Harbor/pool. Scaffold is left unspecified so the reviewed default track is used."],["evidence-2026-07-164","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",80,80,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-163","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",80.3,80.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-617","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.1,57.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-pro:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",53.4,53.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Pro results other than Claude were re-evaluated with Claude Code at temperature 1, top_p 0.95 and 256K context. Claude's value is its officially published score. Qwen says tasks were corrected and baselines re-evaluated. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-swe-bench-pro-v1-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",53.4,53.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: Claude Code harness, temp=1.0, top_p=0.95, 256K context. Opus 4.6 Max uses the officially reported score. Muse Glimmer 51.2 is already stored from the Muse owner card."],["evidence-2026-07-621","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",53.4,53.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-165","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",69.2,69.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2027","claude-sonnet-5","Claude Sonnet 5","As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).",null,"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",63.2,63.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 SWE-Bench Pro comparison score."],["evidence-2026-07-169","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",63.2,63.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-180","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",49.1,49.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-pro:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",56,56,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Pro results other than Claude were re-evaluated with Claude Code at temperature 1, top_p 0.95 and 256K context. Claude's value is its officially published score. Qwen says tasks were corrected and baselines re-evaluated. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-07-178","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",52.1,52.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2049","gemini-3-flash","Gemini 3 Flash","As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.",null,"As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",49.6,49.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","SWE-Bench Pro 49.6% for Gemini 3 Flash from 3.5 Flash-Lite launch comparison."],["evidence-2026-07-011","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",54.2,54.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-001","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",55.1,55.1,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-177","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",55.1,55.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2037","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. SWE-Bench Pro as published in Google launch blog.",null,"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. SWE-Bench Pro as published in Google launch blog.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",54.2,54.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","SWE-Bench Pro 54.2% for 3.5 Flash-Lite."],["evidence-2026-07-1988","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Internal Antigravity harness; self-computed per Google methodology.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Internal Antigravity harness; self-computed per Google methodology.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",58.7,58.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","SWE-Bench Pro from Gemini 3.6 Flash public evaluation table (July 2026)."],["evidence-2026-07-620","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",55.1,55.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-615","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",58.4,58.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-613","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",62.1,62.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-176","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",56.8,56.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-174","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.7,57.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-020","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",58.6,58.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-173","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",58.6,58.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2014","gpt-5-6-luna","GPT-5.6 Luna","As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).",null,"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",62.7,62.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna SWE-Bench Pro comparison score."],["evidence-2026-07-170","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",62.7,62.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-167","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",64.6,64.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-168","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",63.4,63.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2021","grok-4-5","Grok 4.5","As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).",null,"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",64.7,64.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Grok 4.5 SWE-Bench Pro comparison score."],["evidence-2026-07-166","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",64.7,64.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1897","hy3","Hy3","tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.",null,"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.9,57.9,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-06","2026-07-06","production::tencent-hy3-hf","tencent-hy3-hf","Tencent Hy3 model card on Hugging Face","Tencent Hy Team","https://huggingface.co/tencent/Hy3","2026-07-06","2026-07-16","2026-07-16","provider-reported","SWE-bench Pro score listed in Hugging Face evaluation results for tencent/Hy3."],["evidence-2026-07-179","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",50.7,50.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-longcat-2-0-swe-bench-pro-v1","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",59.5,59.5,"percent","higher","2.1.0","ranking-eligible","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score."],["evidence-2026-07-616","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.2,57.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-619","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",56.2,56.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-172","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",59,59,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-622","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",52.4,52.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-614","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",61.5,61.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1910","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (SWE-bench Pro).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (SWE-bench Pro).","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",61.5,61.5,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for SWE-bench Pro; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1910--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (SWE-bench Pro).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (SWE-bench Pro).","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",61.5,61.5,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for SWE-bench Pro; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-08-15-qwen3-6-27b-swe-bench-pro-v1-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",53.5,53.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: Claude Code harness, temp=1.0, top_p=0.95, 256K context. Opus 4.6 Max uses the officially reported score. Muse Glimmer 51.2 is already stored from the Muse owner card."],["evidence-2026-07-618","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",56.6,56.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-171","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",60.6,60.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-175","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.6,57.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-pro:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",55.8,55.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Pro results other than Claude were re-evaluated with Claude Code at temperature 1, top_p 0.95 and 256K context. Claude's value is its officially published score. Qwen says tasks were corrected and baselines re-evaluated. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-swe-bench-pro-v1-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",57.6,57.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: Claude Code harness, temp=1.0, top_p=0.95, 256K context. Opus 4.6 Max uses the officially reported score. Muse Glimmer 51.2 is already stored from the Muse owner card."],["evidence-2026-08-qwen-3-8-max-swe-bench-pro","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",67.7,67.7,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 67.7. The launch post specifies Claude Code, temperature 1, top_p 0.95, 256K context, and refined tasks."],["evidence-2026-08-qwen-3-8-max-swe-bench-pro--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",67.7,67.7,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 67.7. The launch post specifies Claude Code, temperature 1, top_p 0.95, 256K context, and refined tasks."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-pro:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",61.7,61.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Pro results other than Claude were re-evaluated with Claude Code at temperature 1, top_p 0.95 and 256K context. Claude's value is its officially published score. Qwen says tasks were corrected and baselines re-evaluated. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-swe-bench-pro-v1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",61.7,61.7,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: Claude Code harness, temp=1.0, top_p=0.95, 256K context. Opus 4.6 Max uses the officially reported score. Muse Glimmer 51.2 is already stored from the Muse owner card."],["new-model:qwen3-8-flash-next-release-current-2026-08-27:qwen-3-8-flash-next:swe-bench-pro:cell:language:swe-pro:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next public default xhigh configuration","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next public default xhigh configuration","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1",62.5,62.5,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Official Qwen subject row. SWE-Bench Pro v1 was evaluated with Claude Code at temperature 1, top_p 0.95 and 256K context. The existing generic Claude Code protocol is reused; no provider-specific protocol or scoring rule is added."],["evidence-2026-08-muse-glimmer-30b-swe-bench-pro-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1 / original 731-task set",51.2,51.2,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported average pass rate on the original SWE-Bench Pro task set."],["evidence-2026-08-muse-glimmer-30b-swe-bench-pro-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","swe-bench-pro","SWE-Bench Pro","coding","Scale AI","v1 / original 731-task set",51.2,51.2,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported average pass rate on the original SWE-Bench Pro task set."],["benchlm-ref-claude-3-5-sonnet-sweverified-2026-07-21","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49,35.3268,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-sweverified-2026-07-27","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49,35.0829,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-sweverified-2026-08-01","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49,35.0829,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-sweverified-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.7,68.2893,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-sweverified-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.7,67.8177,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-sweverified-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.7,67.8177,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-sweverified-2026-07-21","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.5,70.7928,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-sweverified-2026-07-27","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.5,70.3039,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-sweverified-2026-08-01","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.5,70.3039,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-sweverified-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95,99.3046,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-sweverified-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95,98.6188,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-sweverified-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95,98.6188,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-sweverified-2026-07-21","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.3,69.1238,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-sweverified-2026-07-27","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.3,68.6464,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-sweverified-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.3,68.6464,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-sweverified-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-sweverified-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95.5,99.3094,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-sweverified-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",95.5,99.3094,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-sweverified-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.9,79.694,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-sweverified-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.9,79.1436,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-sweverified-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.9,79.1436,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-sweverified-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.84,79.6106,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-sweverified-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.84,79.0608,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-sweverified-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.84,79.0608,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-sweverified-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",87.6,89.0125,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-sweverified-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",87.6,88.3978,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-sweverified-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",87.6,88.3978,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-sweverified-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",88.6,90.4033,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-sweverified-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",88.6,89.779,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-sweverified-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",88.6,89.779,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-sweverified-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",96,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-sweverified-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",96,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-sweverified-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.548,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-sweverified-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.0331,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-sweverified-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.0331,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-sweverified-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.6,77.886,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-sweverified-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.6,77.3481,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-sweverified-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.6,77.3481,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-sweverified-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85.2,85.6745,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-sweverified-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85.2,85.0829,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-sweverified-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85.2,85.0829,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-sweverified-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",42,25.5911,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-sweverified-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",42,25.4144,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-sweverified-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",42,25.4144,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,76.4951,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,76.4951,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,75.9669,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,75.9669,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,75.9669,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-sweverified-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.6,75.9669,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,77.0515,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,76.5193,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,76.5193,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-sweverified-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.7,69.6801,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-sweverified-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.7,69.1989,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-sweverified-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.7,69.1989,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,77.0515,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,76.5193,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-sweverified-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79,76.5193,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.6078,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.6078,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.0718,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.0718,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.0718,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-sweverified-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",79.4,77.0718,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,79.2768,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,78.7293,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,78.7293,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-sweverified-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.6,69.541,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-sweverified-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.6,69.0608,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-sweverified-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.6,69.0608,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,79.2768,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,78.7293,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-sweverified-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.6,78.7293,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-sweverified-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",63.8,55.911,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-sweverified-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",63.8,55.5249,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-sweverified-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",63.8,55.5249,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-sweverified-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.8,69.8192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-sweverified-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.8,69.337,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-sweverified-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.8,69.337,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverified-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.8,75.3825,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverified-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.8,74.8619,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverified-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.8,74.8619,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-sweverified-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",54.6,43.1154,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-sweverified-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",54.6,42.8177,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-sweverified-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",54.6,42.8177,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-sweverified-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",23.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-sweverified-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",23.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-sweverified-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",23.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-sweverified-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80,78.4423,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-sweverified-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80,77.9006,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-sweverified-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80,77.9006,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-sweverified-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85,85.3964,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-sweverified-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85,84.8066,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-sweverified-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",85,84.8066,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-sweverified-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.7,73.8526,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-sweverified-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.7,73.3425,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-sweverified-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.7,73.3425,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-sweverified-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",70.8,65.6467,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-sweverified-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",70.8,65.1934,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-sweverified-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",70.8,65.1934,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-sweverified-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.4,70.6537,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-sweverified-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.4,70.1657,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-sweverified-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.4,70.1657,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-sweverified-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,75.1043,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-sweverified-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,74.5856,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-sweverified-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,74.5856,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-sweverified-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.2,78.1768,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-sweverified-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.9917,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-sweverified-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.4807,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-sweverified-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.4807,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverified-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.9917,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverified-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.4807,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverified-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.8,73.4807,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-sweverified-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.2,78.7204,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-sweverified-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.2,78.1768,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-sweverified-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.2,78.1768,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-sweverified-2026-07-21","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.6,70.9318,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-sweverified-2026-07-27","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.6,70.442,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-sweverified-2026-08-01","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.6,70.442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-sweverified-2026-07-21","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.9,64.395,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-sweverified-2026-07-27","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.9,63.9503,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-sweverified-2026-08-01","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.9,63.9503,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-sweverified-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.5,69.4019,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-sweverified-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.5,68.9227,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-sweverified-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.5,68.9227,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-sweverified-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,69.2629,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-sweverified-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,68.7845,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-sweverified-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,68.7845,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-sweverified-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.8,71.21,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-sweverified-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.8,70.7182,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-sweverified-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",74.8,70.7182,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-sweverified-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78,75.6606,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-sweverified-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78,75.1381,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-sweverified-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78,75.1381,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-sweverified-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.5,79.1377,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-sweverified-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.5,78.5912,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-sweverified-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.5,78.5912,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,75.1043,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,74.5856,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.6,74.5856,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-sweverified-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.4,74.8261,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-sweverified-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.4,74.3094,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-sweverified-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.4,74.3094,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-sweverified-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",71.9,67.1766,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-sweverified-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",71.9,66.7127,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-sweverified-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",71.9,66.7127,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-sweverified-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49.3,35.7441,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-sweverified-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49.3,35.4972,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-sweverified-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",49.3,35.4972,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-sweverified-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",75.6,72.3227,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-sweverified-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",75.6,71.8232,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-sweverified-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",75.6,71.8232,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-sweverified-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",82.4,81.7803,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-sweverified-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",82.4,81.2155,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-sweverified-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",82.4,81.2155,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-sweverified-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.4,63.6996,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-sweverified-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.4,63.2597,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-sweverified-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.4,63.2597,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-sweverified-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.4,67.872,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-sweverified-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.4,67.4033,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-sweverified-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72.4,67.4033,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-sweverified-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.2,73.1572,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-sweverified-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.2,72.6519,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-sweverified-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",76.2,72.6519,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72,67.3157,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72,66.8508,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",72,66.8508,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.2,63.4214,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.2,62.9834,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",69.2,62.9834,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-sweverified-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.548,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-sweverified-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.0331,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-sweverified-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.2,74.0331,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-sweverified-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.8,76.7733,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-sweverified-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.8,76.2431,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-sweverified-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",78.8,76.2431,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,69.2629,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,68.7845,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",73.4,68.7845,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-sweverified-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.4,78.9986,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-sweverified-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.4,78.453,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-sweverified-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",80.4,78.453,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-sweverified-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.7,75.2434,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-sweverified-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.7,74.7238,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-sweverified-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",77.7,74.7238,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-sweverified-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",53.2,41.1683,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-sweverified-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",53.2,40.884,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-sweverified-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","2024",53.2,40.884,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-182","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",95,95,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-192","claude-haiku-4-5","Claude Haiku 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",73.3,73.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-181","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",95.5,95.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-623","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",80.9,80.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-624","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",80.84,80.84,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-183","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",88.6,88.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-625","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",79.6,79.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-184","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",85.2,85.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-command-a-plus-swe-bench-verified-verified-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",14.4,14.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-swe-bench-verified-verified-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",73.8,73.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-190","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",73.7,73.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-191","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",73.6,73.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-628","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",77.8,77.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-185","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",85,85,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1898","hy3","Hy3","tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.",null,"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",78,78,"percent","higher","1.4.1","reference-only","direct","2026-07-06","2026-07-06","production::tencent-hy3-hf","tencent-hy3-hf","Tencent Hy3 model card on Hugging Face","Tencent Hy Team","https://huggingface.co/tencent/Hy3","2026-07-06","2026-07-16","2026-07-16","provider-reported","SWE-bench Verified resolved rate listed in Hugging Face evaluation results for tencent/Hy3."],["evidence-2026-07-189","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",76.8,76.8,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-laguna-xs-2-1-swe-bench-verified-verified","laguna-xs-2-1","Laguna XS 2.1","Laguna XS 2.1 (thinking enabled)","laguna-xs-2-1-default","Laguna XS 2.1 (thinking enabled)","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",70.9,70.9,"percent","higher","2.1.0","reference-only","direct","2026-07-02","2026-07-02","production::poolside-laguna-xs-2-1-huggingface-2026-08-15","poolside-laguna-xs-2-1-huggingface-2026-08-15","Laguna XS 2.1 model card","Poolside","https://huggingface.co/poolside/Laguna-XS-2.1","2026-07-02","2026-08-15","2026-08-15","provider-reported","Official owner score; mean pass@1 over 4 attempts on Harbor/pool. Scaffold is left unspecified so the reviewed default track is used. Competitor cells were not admitted."],["evidence-2026-07-630","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",74.8,74.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-627","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",78,78,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-15-mimo-v2-5-swe-bench-verified-verified-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",73,73,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-186","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",80.5,80.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-mistral-medium-3-5-swe-bench-verified-verified-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",69.6,69.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-07-629","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",77.4,77.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["nvidia-nemotron-3-5-lightning-swe-bench-verified-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",51.56,51.56,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-626","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",78.8,78.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-187","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",80.4,80.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-188","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",77.7,77.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-15-solar-open-100b-reasoning-swe-bench-verified-verified-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",15.4,15.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-swe-bench-verified-verified","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified",70.4,70.4,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-muse-glimmer-30b-swe-bench-verified-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified / 500 tasks",76,76,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Existing registry classifies SWE-bench Verified as not weighted."],["evidence-2026-08-muse-glimmer-30b-swe-bench-verified-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","swe-bench-verified","SWE-bench Verified","coding","SWE-bench authors","Verified / 500 tasks",76,76,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Existing registry classifies SWE-bench Verified as not weighted."],["benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverifiedarcee-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",72.8,77.4194,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverifiedarcee-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",72.8,77.4194,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-sweverifiedarcee-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",72.8,77.4194,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",70.8,61.2903,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",70.8,61.2903,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",70.8,61.2903,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.4,98.3871,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.4,98.3871,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",75.4,98.3871,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",63.2,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",63.2,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-sweverifiedarcee","SWE-bench Verified (mini-swe-agent-v2)","coding","Arcee AI","2026",63.2,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swerebench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",65.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swerebench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",65.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-swerebench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",65.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-swerebench-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.7,80.5907,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-swerebench-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.7,80.5907,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-swerebench-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.7,80.5907,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swerebench-2026-07-21","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58,69.1983,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swerebench-2026-07-27","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58,69.1983,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swerebench-2026-08-01","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58,69.1983,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-swerebench-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.9,81.4346,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-swerebench-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.9,81.4346,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-swerebench-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",60.9,81.4346,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-swerebench-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",41.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-swerebench-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",41.6,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-swerebench-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",41.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-swerebench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.7,72.1519,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-swerebench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.7,72.1519,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-swerebench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.7,72.1519,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swerebench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.8,89.4515,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swerebench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.8,89.4515,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swerebench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.8,89.4515,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swerebench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.7,89.0295,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swerebench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.7,89.0295,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-swerebench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",62.7,89.0295,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swerebench-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.2,70.0422,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swerebench-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.2,70.0422,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-swerebench-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.2,70.0422,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swerebench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.5,71.308,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swerebench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.5,71.308,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swerebench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.5,71.308,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swerebench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",51.9,43.4599,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swerebench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",51.9,43.4599,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swerebench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",51.9,43.4599,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-swerebench-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.9,72.9958,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-swerebench-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.9,72.9958,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-swerebench-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",58.9,72.9958,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",53.7,51.0549,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",53.7,51.0549,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","swe-rebench","SWE-Rebench","coding","SWE-Rebench","2026",53.7,51.0549,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-783","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,65.3,65.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-786","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,60.7,60.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-790","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,41.6,41.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-784","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,62.8,62.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-785","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,62.7,62.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-788","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,58.2,58.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-787","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,58.5,58.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-789","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","swe-rebench","SWE-Rebench","coding","SWE-Rebench",null,51.9,51.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-15-laguna-xs-2-1-terminal-bench-2-2-0","laguna-xs-2-1","Laguna XS 2.1","Laguna XS 2.1 (thinking enabled)","laguna-xs-2-1-default","Laguna XS 2.1 (thinking enabled)","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2.0",37.5,37.5,"percent","higher","2.1.0","reference-only","direct","2026-07-02","2026-07-02","production::poolside-laguna-xs-2-1-huggingface-2026-08-15","poolside-laguna-xs-2-1-huggingface-2026-08-15","Laguna XS 2.1 model card","Poolside","https://huggingface.co/poolside/Laguna-XS-2.1","2026-07-02","2026-08-15","2026-08-15","provider-reported","Official owner score; mean pass@1 over 5 attempts on Harbor/pool; 48 GB RAM / 32 CPUs. Scaffold is left unspecified so the reviewed default track is used."],["benchlm-ref-claude-fable-terminalbench2-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",84.3,86.4769,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-terminalbench2-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",88,93.0605,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-terminalbench2-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59.3,41.9929,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-terminalbench2-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",65.4,52.847,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-terminalbench2-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",69.4,59.9644,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-terminalbench2-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",74.6,69.2171,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",50,25.4448,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59.1,41.637,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-terminalbench2-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",80.4,79.5374,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-terminalbench2-2026-08-01","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",61.7,46.2633,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-terminalbench2-2026-08-01","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",69.3,59.7865,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.6,37.1886,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-terminalbench2-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.6,37.1886,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.9,37.7224,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-terminalbench2-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",49.1,23.8434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.9,37.7224,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",63.3,49.1103,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-terminalbench2-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",63.3,49.1103,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",67.9,57.2954,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-terminalbench2-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59.1,41.637,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",67.9,57.2954,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-terminalbench2-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",76.2,72.0641,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",54,32.5623,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-terminalbench2-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",41,9.4306,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-terminalbench2-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.2,36.4769,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-terminalbench2-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",63.5,49.4662,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-terminalbench2-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",81,80.605,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-terminalbench2-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",77.3,74.0214,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-terminalbench2-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",75.1,70.1068,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-terminalbench2-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",60,43.2384,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-terminalbench2-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",46.3,18.8612,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-terminalbench2-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",82,82.3843,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-terminalbench2-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",84.7,87.1886,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-terminalbench2-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",91.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-terminalbench2-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",87.4,91.9929,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-terminalbench2-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",47.1,20.2847,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-terminalbench2-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",83.3,84.6975,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-terminalbench2-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",54.4,33.274,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-terminalbench2-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",63.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-terminalbench2-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",64.7,51.6014,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-terminalbench2-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",50.8,26.8683,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-terminalbench2-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",50.8,26.8683,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-terminalbench2-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",66.7,55.1601,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-terminalbench2-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",88.3,93.5943,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-terminalbench2-2026-08-01","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",45.8,17.9715,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-terminalbench2-2026-08-01","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",70.2,61.3879,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-terminalbench2-2026-08-01","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",35.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-terminalbench2-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",46,18.3274,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-terminalbench2-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",65.8,53.5587,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",68.4,58.1851,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-terminalbench2-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",57,37.9004,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-terminalbench2-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",66,53.9146,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-terminalbench2-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59,41.4591,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-terminalbench2-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",80,78.8256,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-terminalbench2-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",56.4,36.8327,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-terminalbench2-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",64.2,50.7117,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-terminalbench2-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",77.5,74.3772,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-terminalbench2-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",43.1,13.1673,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",65.4,52.847,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-terminalbench2-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",41.6,10.4982,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-terminalbench2-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",52.5,29.8932,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",49.4,24.3772,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",40.5,8.5409,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-terminalbench2-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59.3,41.9929,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-terminalbench2-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",61.6,46.0854,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",51.5,28.1139,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-terminalbench2-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",69.7,60.4982,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-terminalbench2-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",70.3,61.5658,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-terminalbench2-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",80.2,79.1815,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",82.1,82.5623,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-terminalbench2-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",59.5,42.3488,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-terminalbench2-2026-08-01","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","terminal-bench-2","Terminal-Bench 2.0","coding","Terminal-Bench contributors","2026",81.5,81.4947,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench21-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-21-provider","Terminal-Bench 2.1 (provider run)","coding","DeepSeek-AI","2026",82.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-terminalbench21-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","terminal-bench-21-provider","Terminal-Bench 2.1 (provider run)","coding","DeepSeek-AI","2026",82.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-07-21","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",11.393,16.0458,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-07-27","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",11.393,16.0458,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-08-01","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",11.393,16.0458,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.63,29.0551,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.63,29.0551,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.63,29.0551,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-vibecodebench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.498,75.3461,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-vibecodebench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.498,75.3461,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-vibecodebench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.498,75.3461,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-vibecodebench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",57.573,81.0853,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-vibecodebench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",57.573,81.0853,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-vibecodebench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",57.573,81.0853,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-vibecodebench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",71.003,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-vibecodebench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",71.003,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-vibecodebench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",71.003,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-07-21","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.621,31.8592,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-07-27","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.621,31.8592,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-08-01","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.621,31.8592,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",51.476,72.4983,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",51.476,72.4983,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",51.476,72.4983,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-07-21","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",5.108,7.1941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-07-27","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",5.108,7.1941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-08-01","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",5.108,7.1941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",49.931,70.3224,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-vibecodebench-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0.4,0.5634,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-vibecodebench-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0.4,0.5634,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-vibecodebench-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0.4,0.5634,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-vibecodebench-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.204,28.4551,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-vibecodebench-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.204,28.4551,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-vibecodebench-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.204,28.4551,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vibecodebench-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.3,20.14,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vibecodebench-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.3,20.14,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vibecodebench-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.3,20.14,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-vibecodebench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",32.034,45.1164,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-vibecodebench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",32.034,45.1164,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-vibecodebench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",32.034,45.1164,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-vibecodebench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",48.683,68.5647,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-vibecodebench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",48.683,68.5647,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-vibecodebench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",48.683,68.5647,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-vibecodebench-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.09,4.3519,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-vibecodebench-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.09,4.3519,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-vibecodebench-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.09,4.3519,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-vibecodebench-2026-07-21","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",23.359,32.8986,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-vibecodebench-2026-07-27","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",23.359,32.8986,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-vibecodebench-2026-08-01","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",23.359,32.8986,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-vibecodebench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",31.456,44.3024,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-vibecodebench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",31.456,44.3024,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-vibecodebench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",31.456,44.3024,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-vibecodebench-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",20.088,28.2918,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-mini-vibecodebench-2026-07-21","gpt-5-mini","GPT-5 mini","Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.171,19.9583,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-mini-vibecodebench-2026-07-27","gpt-5-mini","GPT-5 mini","Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.171,19.9583,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-mini-vibecodebench-2026-08-01","gpt-5-mini","GPT-5 mini","Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.171,19.9583,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-vibecodebench-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",24.606,34.6549,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-vibecodebench-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",24.606,34.6549,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-vibecodebench-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",24.606,34.6549,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-vibecodebench-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",13.115,18.4711,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-vibecodebench-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",13.115,18.4711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-vibecodebench-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",13.115,18.4711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.168,31.2212,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.168,31.2212,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",22.168,31.2212,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vibecodebench-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.499,75.3475,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vibecodebench-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.499,75.3475,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vibecodebench-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",53.499,75.3475,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-vibecodebench-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.912,53.3949,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-vibecodebench-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.912,53.3949,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-vibecodebench-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.912,53.3949,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-vibecodebench-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",61.767,86.9921,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-vibecodebench-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",61.767,86.9921,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-vibecodebench-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",61.767,86.9921,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-vibecodebench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",67.421,94.9551,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-vibecodebench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",67.421,94.9551,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-vibecodebench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",67.421,94.9551,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-vibecodebench-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",47.969,67.5591,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-vibecodebench-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",47.969,67.5591,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-vibecodebench-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",47.969,67.5591,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-vibecodebench-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",26.097,36.7548,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-vibecodebench-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",26.097,36.7548,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-vibecodebench-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",26.097,36.7548,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-vibecodebench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",69.847,98.3719,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-vibecodebench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",69.847,98.3719,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-vibecodebench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",69.847,98.3719,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",1.2,1.6901,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",1.2,1.6901,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",1.2,1.6901,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-vibecodebench-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",4.064,5.7237,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-vibecodebench-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",4.064,5.7237,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-vibecodebench-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",4.064,5.7237,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-vibecodebench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",17.536,24.6975,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-vibecodebench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",17.536,24.6975,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-vibecodebench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",17.536,24.6975,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vibecodebench-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.891,53.3654,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vibecodebench-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.891,53.3654,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vibecodebench-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",37.891,53.3654,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-5-vibecodebench-2026-07-21","minimax-m2-5","MiniMax M2.5","Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.852,20.9174,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-5-vibecodebench-2026-07-27","minimax-m2-5","MiniMax M2.5","Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.852,20.9174,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-5-vibecodebench-2026-08-01","minimax-m2-5","MiniMax M2.5","Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",14.852,20.9174,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibecodebench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",27.037,38.0787,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibecodebench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",27.037,38.0787,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibecodebench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",27.037,38.0787,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-vibecodebench-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",19.674,27.7087,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-vibecodebench-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",19.674,27.7087,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-vibecodebench-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",19.674,27.7087,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-vibecodebench-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.506,4.9378,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-vibecodebench-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.506,4.9378,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-vibecodebench-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",3.506,4.9378,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-vibecodebench-2026-07-21","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",15.738,22.1653,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-vibecodebench-2026-07-27","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",15.738,22.1653,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-vibecodebench-2026-08-01","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",15.738,22.1653,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vibecodebench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",25.564,36.0041,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vibecodebench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",25.564,36.0041,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vibecodebench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vibecodebench","Vibe Code Bench v1.1","coding","Vals AI","2026",25.564,36.0041,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-vibev2-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-vibev2","VIBE V2","coding","MiniMax","2026",50.1,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-vibev2-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-vibev2","VIBE V2","coding","MiniMax","2026",50.1,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-vibev2-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-vibev2","VIBE V2","coding","MiniMax","2026",50.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibepro-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibepro","VIBE-Pro","coding","MiniMax","2026",55.6,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibepro-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibepro","VIBE-Pro","coding","MiniMax","2026",55.6,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-vibepro-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-vibepro","VIBE-Pro","coding","MiniMax","2026",55.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-vulcanbench-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-vulcanbench-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-vulcanbench-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,75.2731,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-vulcanbench-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",78.26,25.0144,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-vulcanbench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-vulcanbench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-vulcanbench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",87,75.2731,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-vulcanbench-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",91.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-vulcanbench-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",91.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-vulcanbench-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",91.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-vulcanbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-vulcanbench","VulcanBench v3","coding","VulcanBench contributors","2026",73.91,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:glm-5-3-flash-2026-08-27:cell:glm-5-3-flash-zai-code-bench:max:1","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Max) as published in Z.AI's comparison table","claude-opus-4-8-max","Claude Opus 4.8 (Max) as published in Z.AI's comparison table","zai-code-bench","Z.ai Code Bench","coding","Z.AI","2026-08",29.5,29.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::zai-glm-5-3-flash-blog-2026-08-26","zai-glm-5-3-flash-blog-2026-08-26","GLM-5.3-Flash official launch and evaluation","Z.AI","https://z.ai/blog/glm-5.3-flash","2026-08-26","2026-08-27","2026-08-27","provider-reported","The chart prints exact Max-effort values of 29.0 for GLM-5.3-Flash and 29.5 for Claude Opus 4.8. Lower-effort plotted points are retained in the image but not numerically inferred. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["evidence-2026-08-15-glm-5-3-zai-code-bench-max","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","zai-code-bench","Z.ai Code Bench","coding","Z.AI","Max",34.5,34.5,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI launch post: Max effort reaches 34.5% at roughly 75K output tokens per task."],["evidence-2026-07-871","mistral-large-3","Mistral Large 3","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","arena-hard-auto","Arena-Hard-Auto","instruction-following","LMSYS Org",null,55.1,55.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-arena-hard","llm-stats-arena-hard","Arena-Hard scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/arena-hard","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["benchlm-ref-claude-3-haiku-aaifbench-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.1,29.1982,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aaifbench-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.1,30.9735,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aaifbench-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.1,30.9735,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaifbench-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",45.4,43.2678,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaifbench-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",45.4,44.6903,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaifbench-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",45.4,44.6903,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.4,58.3964,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.4,59.4395,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.4,59.4395,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaifbench-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.5,70.6505,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaifbench-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.5,71.3864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaifbench-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.5,71.3864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaifbench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,39.6369,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaifbench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaifbench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58,62.3298,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58,63.2743,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58,63.2743,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaifbench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.1,54.9168,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaifbench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.1,56.0472,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaifbench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.1,56.0472,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaifbench-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.6,42.0575,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaifbench-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.6,43.5103,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaifbench-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.6,43.5103,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaifbench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58.6,63.2375,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaifbench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58.6,64.1593,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaifbench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",58.6,64.1593,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaifbench-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43.6,40.5446,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaifbench-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43.6,42.0354,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaifbench-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43.6,42.0354,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaifbench-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",62.2,68.6838,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaifbench-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",62.2,69.469,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaifbench-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",62.2,69.469,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaifbench-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.2,36.9138,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaifbench-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.2,38.4956,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaifbench-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.2,38.4956,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaifbench-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.3843,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaifbench-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.7257,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaifbench-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.7257,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-07-21","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",22.9,9.2284,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-07-27","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",22.9,11.5044,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-08-01","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",22.9,11.5044,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaifbench-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.8,27.2315,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaifbench-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.8,29.056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaifbench-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.8,29.056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaifbench-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,37.3676,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaifbench-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,38.9381,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaifbench-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,38.9381,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaifbench-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.8,31.77,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaifbench-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.8,33.4808,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaifbench-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.8,33.4808,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaifbench-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",49,48.7141,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaifbench-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",49,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaifbench-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",49,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,85.7791,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,85.7791,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaifbench-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.4024,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.5428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.5428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.4024,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.5428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.2,94.5428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.4508,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.4508,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.8909,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.8909,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.8909,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaifbench-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.3,82.8909,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.3177,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.5605,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.5605,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.3177,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.5605,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.5,90.5605,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaifbench-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.6,34.4932,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaifbench-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.6,36.1357,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaifbench-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.6,36.1357,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",25.3,12.8593,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",25.3,15.0442,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",25.3,15.0442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaifbench-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.5,25.2648,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaifbench-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.5,27.1386,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaifbench-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.5,27.1386,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaifbench-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,33.5855,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaifbench-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,35.2507,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaifbench-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,35.2507,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaifbench-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.7,48.2602,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaifbench-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.7,49.5575,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaifbench-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.7,49.5575,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaifbench-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.1,57.9425,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaifbench-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.1,58.9971,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaifbench-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.1,58.9971,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaifbench-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.4,81.0893,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaifbench-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.4,81.5634,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaifbench-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.4,81.5634,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.2,91.3767,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.2,91.5929,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.2,91.5929,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaifbench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.1,91.2254,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaifbench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.1,91.4454,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaifbench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.1,91.4454,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaifbench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.0151,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaifbench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.2655,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaifbench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.2655,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaifbench-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.8,22.6929,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaifbench-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.8,24.6313,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaifbench-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.8,24.6313,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaifbench-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,85.7791,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaifbench-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaifbench-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.5,86.1357,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.4,84.115,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.4,84.5133,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.4,84.5133,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaifbench-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,88.9561,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaifbench-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,89.233,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaifbench-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,89.233,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaifbench-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",35.1,27.6853,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaifbench-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36,30.826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaifbench-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36,30.826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaifbench-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.2,41.4523,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaifbench-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",40.6,37.6106,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaifbench-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",40.6,37.6106,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaifbench-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.6,31.4675,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaifbench-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.6,33.1858,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaifbench-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",37.6,33.1858,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaifbench-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.7,30.1059,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaifbench-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.7,31.8584,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaifbench-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.7,31.8584,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaifbench-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.9,77.3071,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaifbench-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.9,77.8761,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaifbench-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.9,77.8761,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaifbench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.3,83.9637,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaifbench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.3,84.3658,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaifbench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.3,84.3658,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaifbench-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.2,85.3253,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaifbench-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.2,85.6932,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaifbench-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.2,85.6932,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaifbench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.0151,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaifbench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.2655,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaifbench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.3,90.2655,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaifbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.4766,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaifbench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.8407,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaifbench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.8407,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaifbench-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",61.1,67.0197,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaifbench-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",61.1,67.8466,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaifbench-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",61.1,67.8466,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaifbench-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,39.6369,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaifbench-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaifbench-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaifbench-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.3,32.5265,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaifbench-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.3,34.2183,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaifbench-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.3,34.2183,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaifbench-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",32,22.9955,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaifbench-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",32,24.9263,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaifbench-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",32,24.9263,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaifbench-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.3,26.475,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaifbench-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.3,28.3186,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaifbench-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.3,28.3186,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaifbench-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31,21.4826,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaifbench-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31,23.4513,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aaifbench-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31,23.4513,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.174,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.5457,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.5457,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.3918,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.8584,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.8584,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.174,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.5457,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaifbench-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.1,85.5457,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.3918,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.8584,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaifbench-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.6,81.8584,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaifbench-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.9,84.8714,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaifbench-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.9,85.2507,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaifbench-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.9,85.2507,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaifbench-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.4841,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaifbench-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.9735,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaifbench-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.9735,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.4841,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.9735,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70,80.9735,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaifbench-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.6536,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaifbench-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.9381,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaifbench-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.9381,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaifbench-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.6,91.9818,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaifbench-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.6,92.1829,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaifbench-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",77.6,92.1829,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaifbench-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.6536,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaifbench-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.9381,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaifbench-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.4,88.9381,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaifbench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.3843,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaifbench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.7257,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaifbench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.9,86.7257,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaifbench-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.4766,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaifbench-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.8407,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaifbench-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",73.3,85.8407,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaifbench-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.41,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaifbench-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaifbench-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaifbench-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.41,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaifbench-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaifbench-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaifbench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.7,84.5688,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaifbench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.7,84.9558,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaifbench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.7,84.9558,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaifbench-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.2,82.2995,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaifbench-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.2,82.7434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaifbench-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.2,82.7434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaifbench-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",69,78.9713,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaifbench-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",69,79.4985,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaifbench-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",69,79.4985,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaifbench-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",65.1,73.0711,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaifbench-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",65.1,73.7463,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaifbench-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",65.1,73.7463,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaifbench-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",20.5,5.5976,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaifbench-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",20.5,7.9646,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaifbench-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",20.5,7.9646,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaifbench-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",16.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaifbench-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",15.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaifbench-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",15.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaifbench-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",26.2,14.2209,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaifbench-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",25.1,14.7493,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaifbench-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",25.1,14.7493,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaifbench-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",17.6,1.2103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaifbench-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",17.1,2.9499,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaifbench-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",17.1,2.9499,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaifbench-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.7,55.8245,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaifbench-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.7,56.9322,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaifbench-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.7,56.9322,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",50.5,50.9834,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",50.5,52.2124,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",50.5,52.2124,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaifbench-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.5,29.8033,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaifbench-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.5,31.5634,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaifbench-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.5,31.5634,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",52.7,54.3116,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",52.7,55.4572,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",52.7,55.4572,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaifbench-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.3,97.5794,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaifbench-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.3,97.6401,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaifbench-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.3,97.6401,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaifbench-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.4,37.2163,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaifbench-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.4,38.7906,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaifbench-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.4,38.7906,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaifbench-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.7,72.466,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaifbench-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.7,73.1563,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaifbench-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.7,73.1563,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaifbench-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,37.3676,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaifbench-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,38.9381,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaifbench-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",41.5,38.9381,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaifbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,80.7867,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaifbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,81.2684,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaifbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,81.2684,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaifbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,80.7867,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaifbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,81.2684,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaifbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.2,81.2684,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaifbench-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76,89.5613,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaifbench-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76,89.823,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaifbench-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76,89.823,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaifbench-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.1,70.0454,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaifbench-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.1,70.7965,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaifbench-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.1,70.7965,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",55.6,58.6989,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.3,56.3422,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.3,56.3422,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.1,24.6596,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.1,26.5487,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.1,26.5487,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaifbench-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",57.4,61.4221,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaifbench-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",57.4,62.3894,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaifbench-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",57.4,62.3894,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaifbench-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,33.5855,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaifbench-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,35.2507,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaifbench-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39,35.2507,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaifbench-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,39.6369,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaifbench-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaifbench-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",43,41.1504,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaifbench-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.5,34.3419,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaifbench-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.5,35.9882,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaifbench-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.5,35.9882,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaifbench-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.9,34.947,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaifbench-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.9,36.5782,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaifbench-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.9,36.5782,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaifbench-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.5,55.5219,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaifbench-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.5,56.6372,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaifbench-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",53.5,56.6372,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaifbench-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,78.6687,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaifbench-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,79.2035,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaifbench-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,79.2035,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaifbench-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.9,95.4614,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaifbench-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.9,95.5752,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaifbench-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",79.9,95.5752,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaifbench-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.1074,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaifbench-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.3805,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaifbench-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.3805,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaifbench-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",82.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaifbench-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",82.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaifbench-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",82.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaifbench-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.2,21.7852,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaifbench-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.2,23.7463,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaifbench-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",31.2,23.7463,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaifbench-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.2,29.3495,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaifbench-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.2,31.1209,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaifbench-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",36.2,31.1209,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaifbench-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.3,34.0393,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaifbench-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.3,35.6932,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaifbench-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",39.3,35.6932,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,78.6687,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,79.2035,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",68.8,79.2035,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaifbench-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,47.5038,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaifbench-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,48.8201,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaifbench-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,48.8201,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaifbench-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,47.5038,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaifbench-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,48.8201,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaifbench-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",48.2,48.8201,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaifbench-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.41,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaifbench-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaifbench-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.9,89.6755,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.1,82.1483,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.1,82.5959,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.1,82.5959,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.2,70.1967,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.2,70.944,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",63.2,70.944,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaifbench-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.4,97.7307,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaifbench-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.4,97.7876,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaifbench-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",81.4,97.7876,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaifbench-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.2,32.3752,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaifbench-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.2,34.0708,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaifbench-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.2,34.0708,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaifbench-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.1,32.2239,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaifbench-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.1,33.9233,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaifbench-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",38.1,33.9233,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaifbench-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.3,80.938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaifbench-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.3,81.4159,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaifbench-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",70.3,81.4159,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaifbench-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.4,82.6021,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaifbench-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.4,83.0383,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaifbench-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",71.4,83.0383,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaifbench-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",23.5,10.1362,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaifbench-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",23.5,12.3894,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaifbench-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",23.5,12.3894,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaifbench-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.6,90.469,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaifbench-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.6,90.708,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaifbench-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",76.6,90.708,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaifbench-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.1,41.3011,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaifbench-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.1,42.7729,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaifbench-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",44.1,42.7729,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaifbench-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,88.9561,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaifbench-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,89.233,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaifbench-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.6,89.233,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaifbench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.7973,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaifbench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.9528,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaifbench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.9528,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaifbench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.7973,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaifbench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.9528,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaifbench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78.8,93.9528,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.1074,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.3805,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.7,89.3805,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.5,84.2663,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.5,84.6608,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",72.5,84.6608,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaifbench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.6,76.8533,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaifbench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.6,77.4336,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaifbench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.6,77.4336,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaifbench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.2,88.351,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaifbench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.2,88.6431,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaifbench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",75.2,88.6431,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.4,72.0121,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.4,72.7139,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",64.4,72.7139,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaifbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",80.5,96.3691,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaifbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",80.5,96.4602,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaifbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",80.5,96.4602,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaifbench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78,92.587,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaifbench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78,92.7729,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaifbench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",78,92.7729,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaifbench-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.4,26.6263,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaifbench-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.4,28.4661,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaifbench-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",34.4,28.4661,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaifbench-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",26.5,14.6747,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaifbench-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",26.5,16.8142,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaifbench-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",26.5,16.8142,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaifbench-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.7,25.5673,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaifbench-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.7,27.4336,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaifbench-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",33.7,27.4336,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaifbench-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.3,76.3994,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaifbench-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.3,76.9912,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaifbench-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",67.3,76.9912,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaifbench-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,59.7579,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaifbench-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,60.767,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaifbench-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,60.767,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaifbench-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,59.7579,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaifbench-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,60.767,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaifbench-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","ifbench","IFBench","instruction-following","IFBench","2026",56.3,60.767,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ifbench:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","ifbench","IFBench","instruction-following","IFBench","294 tasks",62.5,62.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-ifbench-294-tasks-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","ifbench","IFBench","instruction-following","IFBench","294 tasks",62.5,62.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ifbench:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","ifbench","IFBench","instruction-following","IFBench","294 tasks",79.2,79.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-muse-glimmer-30b-ifbench-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","ifbench","IFBench","instruction-following","IFBench","294 tasks",77,77,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Existing registry classifies IFBench as not weighted."],["evidence-2026-08-muse-glimmer-30b-ifbench-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","ifbench","IFBench","instruction-following","IFBench","294 tasks",77,77,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Existing registry classifies IFBench as not weighted."],["evidence-2026-08-15-qwen3-6-27b-ifbench-294-tasks-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","ifbench","IFBench","instruction-following","IFBench","294 tasks",69.1,69.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ifbench:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","ifbench","IFBench","instruction-following","IFBench","294 tasks",79.1,79.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-ifbench-294-tasks-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","ifbench","IFBench","instruction-following","IFBench","294 tasks",79.1,79.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ifbench:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","ifbench","IFBench","instruction-following","IFBench","294 tasks",79.5,79.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-ifbench-294-tasks","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","ifbench","IFBench","instruction-following","IFBench","294 tasks",79.5,79.5,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:ifbench:cell:language:ifbench:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","ifbench","IFBench","instruction-following","IFBench","294 tasks",81.3,81.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["evidence-2026-08-15-command-a-plus-ifbench-standard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","ifbench","IFBench","instruction-following","IFBench","standard",73.9,73.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-ifbench-standard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","ifbench","IFBench","instruction-following","IFBench","standard",80.3,80.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-ifbench-standard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","ifbench","IFBench","instruction-following","IFBench","standard",67.1,67.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-ifbench-standard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","ifbench","IFBench","instruction-following","IFBench","standard",69,69,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-ifbench-standard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","ifbench","IFBench","instruction-following","IFBench","standard",57.7,57.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-ifbench-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","ifbench","IFBench","instruction-following","IFBench","standard",80,80,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-832","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,63.5,63.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1407","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,53.7415,53.7415,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1397","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,55.4422,55.4422,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-810","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,58,58,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-838","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,44.6,44.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-839","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,43.6,43.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-833","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,62.2,62.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1779","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,48.2993,48.2993,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1730","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,54.6939,54.6939,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1384","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,57.2789,57.2789,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-840","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,41.2,41.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1636","command-a-plus","Command A+","Command A+",null,"Command A+","ifbench","IFBench","instruction-following","IFBench",null,73.9456,73.9456,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1543","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,57.0068,57.0068,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1529","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,60.6803,60.6803,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1842","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,52.3129,52.3129,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1804","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","ifbench","IFBench","instruction-following","IFBench",null,48.7075,48.7075,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-835","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,55.1,55.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-829","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,70.4,70.4,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-813","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,77.2,77.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-814","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,77.1,77.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-808","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,76.3,76.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1884","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,73.5374,73.5374,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1623","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,72.449,72.449,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-820","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.6,75.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1744","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,43.4014,43.4014,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1557","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,67.8912,67.8912,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-827","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,72.3,72.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-824","glm-5-turbo","GLM-5-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,73.2,73.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-815","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,76.3,76.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-823","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,73.3,73.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-834","glm-5v-turbo","GLM-5V-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,61.1,61.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1344","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","ifbench","IFBench","instruction-following","IFBench",null,73.0612,73.0612,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1344--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","ifbench","IFBench","instruction-following","IFBench",null,73.0612,73.0612,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1329","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","ifbench","IFBench","instruction-following","IFBench",null,74.1497,74.1497,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1329--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","ifbench","IFBench","instruction-following","IFBench",null,74.1497,74.1497,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1356","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","ifbench","IFBench","instruction-following","IFBench",null,71.1565,71.1565,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1356--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","ifbench","IFBench","instruction-following","IFBench",null,71.1565,71.1565,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-825","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,72.9,72.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1307","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,75.4422,75.4422,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1307--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,75.4422,75.4422,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1317","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,77.619,77.619,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1317--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,77.619,77.619,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-821","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.4,75.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-822","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,73.9,73.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1296","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,73.2653,73.2653,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1296--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","ifbench","IFBench","instruction-following","IFBench",null,73.2653,73.2653,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-818","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.9,75.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-816","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.9,75.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-826","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,72.7,72.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-828","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,71.2,71.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1874","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","ifbench","IFBench","instruction-following","IFBench",null,45.8503,45.8503,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1663","grok-4","Grok 4","Grok 4",null,"Grok 4","ifbench","IFBench","instruction-following","IFBench",null,53.6735,53.6735,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1767","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,50.5442,50.5442,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1450","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,81.2245,81.2245,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-805","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,81.3,81.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1896","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","ifbench","IFBench","instruction-following","IFBench",null,41.3605,41.3605,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1853","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","ifbench","IFBench","instruction-following","IFBench",null,41.7007,41.7007,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1674","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","ifbench","IFBench","instruction-following","IFBench",null,68.0952,68.0952,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-830","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,70.2,70.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1424","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","ifbench","IFBench","instruction-following","IFBench",null,75.9864,75.9864,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1439","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","ifbench","IFBench","instruction-following","IFBench",null,63.1293,63.1293,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1789","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","ifbench","IFBench","instruction-following","IFBench",null,56.8707,56.8707,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1593","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,64.2177,64.2177,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-836","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,53.5,53.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-831","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,68.8,68.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1579","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","ifbench","IFBench","instruction-following","IFBench",null,67.1429,67.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-812","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,79.9,79.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1755","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","ifbench","IFBench","instruction-following","IFBench",null,72.3129,72.3129,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1694","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","ifbench","IFBench","instruction-following","IFBench",null,69.8639,69.8639,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1566","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","ifbench","IFBench","instruction-following","IFBench",null,71.6327,71.6327,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-819","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.7,75.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-811","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,82.9,82.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-841","mistral-large-3","Mistral Large 3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,36.2,36.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-837","mistral-small-4","Mistral Small 4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,48.2,48.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-817","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.9,75.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1830","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,71.4966,71.4966,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1608","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,81.3605,81.3605,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1651","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","ifbench","IFBench","instruction-following","IFBench",null,79.0476,79.0476,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-ifbench-loose-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","ifbench","IFBench","instruction-following","IFBench",null,71.88,71.88,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1863","o1","o1","o1",null,"o1","ifbench","IFBench","instruction-following","IFBench",null,70.3401,70.3401,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1369","o3","o3","o3",null,"o3","ifbench","IFBench","instruction-following","IFBench",null,71.4286,71.4286,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1816","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","ifbench","IFBench","instruction-following","IFBench",null,68.7075,68.7075,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1816--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","ifbench","IFBench","instruction-following","IFBench",null,68.7075,68.7075,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1683","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","ifbench","IFBench","instruction-following","IFBench",null,70.7483,70.7483,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1503","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,75.7143,75.7143,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1514","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,75.5782,75.5782,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1715","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,72.517,72.517,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1489","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,78.7755,78.7755,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1704","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","ifbench","IFBench","instruction-following","IFBench",null,51.1565,51.1565,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1463","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","ifbench","IFBench","instruction-following","IFBench",null,67.551,67.551,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1472","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","ifbench","IFBench","instruction-following","IFBench",null,76.5986,76.5986,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-809","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,75.8,75.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-806","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,80.5,80.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-807","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifbench","IFBench","instruction-following","IFBench",null,79.1,79.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-qwen-3-8-max-ifbench","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","ifbench","IFBench","instruction-following","IFBench",null,82.8,82.8,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 82.8 on IFBench."],["evidence-2026-08-qwen-3-8-max-ifbench--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","ifbench","IFBench","instruction-following","IFBench",null,82.8,82.8,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 82.8 on IFBench."],["evidence-2026-08-15-longcat-2-0-ifeval-standard","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","ifeval","IFEval","instruction-following","Google Research","standard",90,90,"percent","higher","2.1.0","reference-only","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score. IFEval is not weighted."],["evidence-2026-07-337","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,63.5,63.5,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-717","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,90.9,90.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-732","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,44.6,44.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-733","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,43.6,43.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-338","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,62.2,62.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-734","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,41.2,41.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-730","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,55.1,55.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-727","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,70.4,70.4,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-330","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,77.2,77.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-331","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,77.1,77.1,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-328","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,76.3,76.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-723","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.6,75.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-716","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,92.6,92.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-725","glm-5-turbo","GLM-5-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,73.2,73.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-719","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,76.3,76.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-724","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,73.3,73.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-729","glm-5v-turbo","GLM-5V-Turbo","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,61.1,61.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-726","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,72.9,72.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-333","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.4,75.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-334","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,73.9,73.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-721","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.9,75.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-332","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.9,75.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-335","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,72.7,72.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-336","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,71.2,71.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-327","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,81.3,81.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-295","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,93.9,93.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-731","mimo-v2-omni","MiMo-V2-Omni","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,53.5,53.5,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-728","mimo-v2-pro","MiMo-V2-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,68.8,68.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-718","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,79.9,79.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-722","minimax-m2-7","MiniMax M2.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.7,75.7,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-329","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,82.9,82.9,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-340","mistral-large-3","Mistral Large 3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,36.2,36.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-339","mistral-small-4","Mistral Small 4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,48.2,48.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-720","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,75.9,75.9,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-715","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,94.3,94.3,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-294","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,94.3,94.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-293","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","ifeval","IFEval","instruction-following","Google Research",null,94.6,94.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["benchlm-ref-claude-opus-4-5-ifbench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",58,42.0601,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ifbench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",58,42.0601,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ifbench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",58,42.0601,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-ifbench-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",76.3,81.3305,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-ifbench-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",76.3,81.3305,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-ifbench-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",76.3,81.3305,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-ifbench-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.3,92.0601,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-ifbench-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.3,92.0601,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-ifbench-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.3,92.0601,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-ifbench-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",63.1,53.0043,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-ifbench-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",63.1,53.0043,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-ifbench-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",63.1,53.0043,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-ifbench-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.8,88.8412,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-ifbench-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.8,88.8412,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-ifbench-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.8,88.8412,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-ifbench-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",82.2,93.9914,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifbench-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",38.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifbench-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",38.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifbench-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",38.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",56.47,38.7768,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",56.47,38.7768,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",56.47,38.7768,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-ifbench-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",57,39.9142,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-ifbench-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",57,39.9142,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-ifbench-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",57,39.9142,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-ifbench-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",85,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-ifbench-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",85,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-ifbench-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",85,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifbench-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",46.67,17.7468,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifbench-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",46.67,17.7468,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifbench-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",46.67,17.7468,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",74.2,76.824,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",74.2,76.824,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",74.2,76.824,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-ifbench-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.7,92.9185,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-ifbench-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.7,92.9185,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-ifbench-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",81.7,92.9185,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifbench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",75.8,80.2575,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifbench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",75.8,80.2575,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifbench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",75.8,80.2575,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifbench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifbench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifbench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",79.1,87.3391,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifbench-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",52.56,30.3863,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifbench-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",52.56,30.3863,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifbench-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifbench","Instruction Following Benchmark","instruction-following","BenchLM registry","2025",52.56,30.3863,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-ifeval-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.82,99.4681,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-ifeval-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.82,99.4681,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-ifeval-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.82,99.4681,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ifeval-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",90.9,87.8842,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ifeval-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",90.9,87.8842,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ifeval-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",90.9,87.8842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-ifeval-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",86.1,73.6998,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-ifeval-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",86.1,73.6998,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-ifeval-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",86.1,73.6998,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-ifeval-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-ifeval-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-ifeval-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-ifeval-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",87.4,77.5414,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-ifeval-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",87.4,77.5414,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-ifeval-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",87.4,77.5414,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-ifeval-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",88.5,80.792,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-ifeval-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",88.5,80.792,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-ifeval-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",88.5,80.792,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-ifeval-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",83.2,65.13,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-ifeval-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",83.2,65.13,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-ifeval-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",83.2,65.13,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-ifeval-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-ifeval-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-ifeval-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifeval-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",71.71,31.1761,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifeval-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",71.71,31.1761,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-ifeval-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",71.71,31.1761,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.84,90.6619,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.84,90.6619,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.84,90.6619,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-ifeval-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",61.16,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-ifeval-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",61.16,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-ifeval-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",61.16,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-07-21","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",75.8,43.2624,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-07-27","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",75.8,43.2624,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",75.8,43.2624,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-07-21","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",76.5,45.331,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-07-27","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",76.5,45.331,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",76.5,45.331,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifeval-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",80.41,56.8853,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifeval-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",80.41,56.8853,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-ifeval-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",80.41,56.8853,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-ifeval-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.2,91.7258,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-ifeval-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.2,91.7258,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-ifeval-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.2,91.7258,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-ifeval-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-ifeval-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-ifeval-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.9,96.7494,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-ifeval-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",95,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-ifeval-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",95,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-ifeval-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",95,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ifeval-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ifeval-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ifeval-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",92.6,92.9078,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.4,95.2719,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.4,95.2719,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",93.4,95.2719,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.9,90.8392,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.9,90.8392,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",91.9,90.8392,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifeval-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifeval-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ifeval-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifeval-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifeval-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-ifeval-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.3,97.9314,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifeval-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.6,98.818,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifeval-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.6,98.818,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ifeval-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",94.6,98.818,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifeval-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",85.58,72.1631,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifeval-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",85.58,72.1631,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-ifeval-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-ifeval","Instruction-Following Eval","instruction-following","Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","2023",85.58,72.1631,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-sobvalueacc-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-sobvalueacc","Structured Output Benchmark Value Accuracy","instruction-following","Interfaze","2026",79.5,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-sobvalueacc-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-sobvalueacc","Structured Output Benchmark Value Accuracy","instruction-following","Interfaze","2026",79.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-sobvalueacc-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-sobvalueacc","Structured Output Benchmark Value Accuracy","instruction-following","Interfaze","2026",79.5,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-acecyberrangesolved-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-acecyberrangesolved","ACE Cyber Range Challenges Solved","knowledge","NIST CAISI and UK AISI","2026",0,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-acecyberrangesolved-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-acecyberrangesolved","ACE Cyber Range Challenges Solved","knowledge","NIST CAISI and UK AISI","2026",0,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-agentslastexam-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","2026",25.2,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-agentslastexam-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","2026",25.2,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-flash-0731-agents-last-exam-2026-07-31","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 official model card; exact ALE configuration not specified",null,"DeepSeek-V4-Flash-0731 official model card; exact ALE configuration not specified","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","2026",25.2,25.2,"percent","higher","2.3.0","reference-only","direct","2026-07-31","2026-07-31","production::deepseek-v4-flash-0731-model-card","deepseek-v4-flash-0731-model-card","DeepSeek-V4-Flash-0731 official model card and provider evaluation","DeepSeek","https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","2026-07-31","2026-08-24","2026-08-24","provider-reported","Source-native DeepSeek subject value. The model card does not state that the code-agent harness footnote applies to Agents' Last Exam, so absolute configuration remains unresolved."],["deepseek-v4-flash-vision-exp-agents-last-exam-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","Provider chart does not specify an exact evaluation system",null,"Provider chart does not specify an exact evaluation system","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","2026",27.3,27.3,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value; exact evaluation system, harness and model configuration are unavailable in the provider chart."],["new-model:aa-glm-5-3-flash-2026-08-27:glm-5-3-flash:benchlm-alebench:cell:glm-5-3-flash-launch:ale:0","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max effort, documented default)","glm-5-3-flash-max","GLM-5.3-Flash (max effort, documented default)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","2026",26.3,26.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-26","2026-08-27","2026-08-27","source-checked","Official protocol with Claude Code, max effort, 1M context, 64K output and Tool Search disabled."],["evidence-2026-08-15-claude-fable-5-benchlm-alebench-ale-cli-table","claude-fable-5","Claude Fable 5","Claude Fable 5 w/ fallback as published by Z.AI","claude-fable-5-max","Claude Fable 5 w/ fallback as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",23.8,23.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-claude-opus-4-8-benchlm-alebench-ale-cli-table","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 as published by Z.AI","claude-opus-4-8-max","Claude Opus 4.8 as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",25.7,25.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-deepseek-v4-pro-0813-benchlm-alebench-ale-cli-table","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 as published by Z.AI","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",25.7,25.7,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-glm-5-2-benchlm-alebench-ale-cli-table","glm-5-2","GLM-5.2","GLM-5.2 as published by Z.AI","glm-5-2-max","GLM-5.2 as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",23.8,23.8,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-glm-5-3-benchlm-alebench-ale-cli","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",28.5,28.5,"percent","higher","2.1.0","reference-only","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-gpt-5-6-sol-benchlm-alebench-ale-cli-table","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol as published by Z.AI","gpt-5-6-sol-max","GPT-5.6 Sol as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",28.6,28.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-kimi-k3-benchlm-alebench-ale-cli-table","kimi-k3","Kimi K3","Kimi K3 as published by Z.AI","kimi-k3-max","Kimi K3 as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",27.6,27.6,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["evidence-2026-08-15-qwen-3-8-max-benchlm-alebench-ale-cli-table","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max as published by Z.AI","qwen-3-8-max-xhigh","Qwen3.8 Max as published by Z.AI","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","ALE-CLI",27,27,"percent","higher","2.1.0","excluded","direct","2026-08-14","2026-08-14","production::zai-glm-5-3-blog-2026-08-14","zai-glm-5-3-blog-2026-08-14","GLM-5.3: Frontier Coding with Emergent Cyber Capabilities","Z.AI","https://z.ai/blog/glm-5.3","2026-08-14","2026-08-15","2026-08-15","provider-reported","Official Z.AI footnote for GLM-5.3: official ALE protocol, 1M context, 64K max output, 105 tasks."],["deepseek-v4-pro-0813-release-agents-last-exam-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",25.7,25.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["google-gemini-37-eval-agents-last-exam-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",33.3,33.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["deepseek-v4-pro-0813-release-agents-last-exam-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",15.8,15.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-agents-last-exam-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",25.2,25.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-agents-last-exam-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",16.5,16.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-agents-last-exam-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",25.7,25.7,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["google-gemini-37-eval-agents-last-exam-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",24.2,24.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-agents-last-exam-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",26.3,26.3,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["deepseek-v4-pro-0813-release-agents-last-exam-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",23.8,23.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["google-gemini-37-eval-agents-last-exam-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",28,28,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["deepseek-v4-pro-0813-release-agents-last-exam-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","August 2026",27.6,27.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ale-pass1:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",25.2,25.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen3-6-27b-benchlm-alebench-pass-1-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",10.6,10.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ale-pass1:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",13.2,13.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-alebench-pass-1-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",13.2,13.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ale-pass1:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",20.4,20.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-alebench-pass-1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",20.4,20.4,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-alebench:cell:language:ale-pass1:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Pass@1",24.3,24.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","pass@1 Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["evidence-2026-08-15-qwen3-6-27b-benchlm-alebench-score-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",27.3,27.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ale-score:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",33.6,33.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-alebench-score-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",33.6,33.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:ale-score:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",42.9,42.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-alebench-score","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",42.9,42.9,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-alebench:cell:language:ale-score:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-alebench","Agents Last Exam","knowledge","UC Berkeley RDI","Score",51.2,51.2,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-deepseek-v4-flash-base-agieval-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",82.6,96.9136,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-agieval-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",82.6,96.9136,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-agieval-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",82.6,96.9136,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-agieval-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",83.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-agieval-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",83.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-agieval-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",83.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-agieval-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",66.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-agieval-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",66.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-agieval-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-agieval","AGIEval","knowledge","DeepSeek-AI","2026",66.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-organicchemistryv2-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-organicchemistryv2","Anthropic Organic Chemistry V2 evaluation","knowledge","Anthropic","2026",61.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-organicchemistryv2-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-organicchemistryv2","Anthropic Organic Chemistry V2 evaluation","knowledge","Anthropic","2026",61.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-proteindesign-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-proteindesign","Anthropic Protein Design evaluation","knowledge","Anthropic","2026",42.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-proteindesign-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-proteindesign","Anthropic Protein Design evaluation","knowledge","Anthropic","2026",42.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",91.3,81.7308,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",91.3,81.7308,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",91.3,81.7308,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaglobalmmlulite-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaglobalmmlulite-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaglobalmmlulite-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",92.2,90.3846,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",93.2,100,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",93.2,100,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",93.2,100,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",82.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",82.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaglobalmmlulite","Artificial Analysis Global-MMLU-Lite","knowledge","Artificial Analysis","2026",82.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammlupro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",88.9,90,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammlupro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",88.9,90,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammlupro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",88.9,90,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.5,96.6667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.5,96.6667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.5,96.6667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammlupro-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.8,100,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammlupro-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.8,100,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammlupro-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",89.8,100,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aammlupro-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",80.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aammlupro-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",80.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aammlupro-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aammlupro","Artificial Analysis MMLU-Pro","knowledge","Artificial Analysis","2026",80.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.2,24.055,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.2,24.055,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.2,24.055,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.4,32.9897,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.4,32.9897,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.4,32.9897,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omniscienceaccuracy-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",61.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omniscienceaccuracy-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",61.4,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omniscienceaccuracy-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",61.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.7,73.0241,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.7,73.0241,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.7,73.0241,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omniscienceaccuracy-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.4,74.2268,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omniscienceaccuracy-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.4,74.2268,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omniscienceaccuracy-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.4,74.2268,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.2,72.1649,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.2,72.1649,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.2,72.1649,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omniscienceaccuracy-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.8,73.1959,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omniscienceaccuracy-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.8,73.1959,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omniscienceaccuracy-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.8,73.1959,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.5,69.244,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.5,69.244,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.5,69.244,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.6,74.5704,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.6,74.5704,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46.6,74.5704,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-omniscienceaccuracy-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",54.2,87.6289,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-omniscienceaccuracy-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",54.2,87.6289,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38,59.7938,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38,59.7938,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38,59.7938,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.3,60.3093,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.3,60.3093,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.3,60.3093,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omniscienceaccuracy-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.9,9.7938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omniscienceaccuracy-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.9,9.7938,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omniscienceaccuracy-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.9,9.7938,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omniscienceaccuracy-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",28.8,43.9863,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omniscienceaccuracy-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",28.8,43.9863,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omniscienceaccuracy-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",28.8,43.9863,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.1,34.1924,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.1,34.1924,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.1,34.1924,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omniscienceaccuracy-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",35.5,55.4983,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.2,58.4192,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omniscienceaccuracy-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.8,66.323,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.3,68.9003,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31,47.7663,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31,47.7663,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31,47.7663,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",4.7,2.5773,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",4.7,2.5773,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",4.7,2.5773,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",10.4,12.3711,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",10.4,12.3711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",10.4,12.3711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.5,40.0344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.5,40.0344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.5,40.0344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39,61.512,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39,61.512,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39,61.512,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.5,72.6804,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.5,72.6804,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.5,72.6804,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.9,90.5498,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.9,90.5498,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.9,90.5498,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",36.4,57.0447,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",36.4,57.0447,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",36.4,57.0447,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.3,89.5189,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.3,89.5189,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",55.3,89.5189,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.9,83.677,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.9,83.677,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.9,83.677,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.3,46.5636,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.3,46.5636,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.3,46.5636,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50.2,80.756,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50.2,80.756,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50.2,80.756,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.5,15.9794,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.5,15.9794,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.5,15.9794,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16,21.9931,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16,21.9931,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16,21.9931,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.2,25.7732,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.2,25.7732,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.2,25.7732,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.7,6.0137,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.7,6.0137,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.7,6.0137,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.6,9.2784,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.6,9.2784,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",8.6,9.2784,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omniscienceaccuracy-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.8,30.2405,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omniscienceaccuracy-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.8,30.2405,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omniscienceaccuracy-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.8,30.2405,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omniscienceaccuracy-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.3,44.8454,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omniscienceaccuracy-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.3,44.8454,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omniscienceaccuracy-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.3,44.8454,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omniscienceaccuracy-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.9,40.7216,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omniscienceaccuracy-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.9,40.7216,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omniscienceaccuracy-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.9,40.7216,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29,44.3299,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29,44.3299,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29,44.3299,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omniscienceaccuracy-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omniscienceaccuracy-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omniscienceaccuracy-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omniscienceaccuracy-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omniscienceaccuracy-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omniscienceaccuracy-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.1,44.5017,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.1,44.5017,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",29.1,44.5017,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.2,36.0825,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.5,24.5704,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.5,24.5704,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.5,24.5704,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.3,17.354,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.3,17.354,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.3,17.354,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omniscienceaccuracy-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.7,28.3505,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omniscienceaccuracy-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.7,28.3505,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omniscienceaccuracy-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.7,28.3505,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.9,61.3402,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.6,59.1065,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.6,59.1065,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.6,59.1065,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",39.2,61.8557,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.8,69.7595,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.8,69.7595,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",43.8,69.7595,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.7,64.433,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.8,83.5052,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.8,83.5052,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",51.8,83.5052,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50,80.4124,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50,80.4124,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",50,80.4124,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.5,58.9347,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.5,58.9347,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.5,58.9347,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",56.9,92.268,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",56.9,92.268,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",56.9,92.268,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.5,65.8076,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.5,65.8076,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.5,65.8076,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",58.5,95.0172,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",58.5,95.0172,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",58.5,95.0172,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.9,73.3677,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.9,73.3677,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",45.9,73.3677,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.5,31.4433,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.5,31.4433,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.5,31.4433,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.5,21.134,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.1,4.9828,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.1,4.9828,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",6.1,4.9828,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.3,3.6082,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.3,3.6082,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.3,3.6082,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.7,0.8591,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.7,0.8591,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",3.7,0.8591,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omniscienceaccuracy-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.4,65.6357,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omniscienceaccuracy-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.4,65.6357,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omniscienceaccuracy-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",41.4,65.6357,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.3,37.9725,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.3,37.9725,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.3,37.9725,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omniscienceaccuracy-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.6,53.9519,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omniscienceaccuracy-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.6,53.9519,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omniscienceaccuracy-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.6,53.9519,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omniscienceaccuracy-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",52.1,84.0206,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omniscienceaccuracy-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",52.1,84.0206,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omniscienceaccuracy-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",52.1,84.0206,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.8,35.3952,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.8,35.3952,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",23.8,35.3952,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omniscienceaccuracy-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omniscienceaccuracy-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omniscienceaccuracy-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omniscienceaccuracy-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omniscienceaccuracy-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omniscienceaccuracy-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.5,48.6254,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omniscienceaccuracy-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40,63.2302,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omniscienceaccuracy-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40,63.2302,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omniscienceaccuracy-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40,63.2302,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omniscienceaccuracy-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16.5,22.8522,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omniscienceaccuracy-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16.5,22.8522,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omniscienceaccuracy-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",16.5,22.8522,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omniscienceaccuracy-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omniscienceaccuracy-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omniscienceaccuracy-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omniscienceaccuracy-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omniscienceaccuracy-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omniscienceaccuracy-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.3,53.4364,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",32.8,50.8591,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",32.8,50.8591,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",32.8,50.8591,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.6,60.8247,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.6,60.8247,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.6,60.8247,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omniscienceaccuracy-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46,73.5395,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omniscienceaccuracy-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46,73.5395,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omniscienceaccuracy-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",46,73.5395,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",9.4,10.6529,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",9.4,10.6529,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",9.4,10.6529,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.2,3.4364,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.2,3.4364,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",5.2,3.4364,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.4,20.9622,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.4,20.9622,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.4,20.9622,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.3,32.8179,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.3,32.8179,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.3,32.8179,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.3,36.2543,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.3,36.2543,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.3,36.2543,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.6,19.5876,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.6,19.5876,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.6,19.5876,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.2,20.6186,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.2,20.6186,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.2,20.6186,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.7,26.6323,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.7,26.6323,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.7,26.6323,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.8,40.5498,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.6,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.1,39.3471,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.1,39.3471,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.1,39.3471,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omniscienceaccuracy-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15,20.2749,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omniscienceaccuracy-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15,20.2749,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omniscienceaccuracy-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15,20.2749,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.1,29.0378,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.1,29.0378,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.1,29.0378,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.1,35.9107,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.1,35.9107,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.1,35.9107,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.3,25.945,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.3,25.945,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.3,25.945,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.1,37.6289,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omniscienceaccuracy-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omniscienceaccuracy-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omniscienceaccuracy-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.1,32.4742,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omniscienceaccuracy-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",44.6,71.134,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omniscienceaccuracy-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",44.6,71.134,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omniscienceaccuracy-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",44.6,71.134,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.6,64.2612,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.6,64.2612,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",40.6,64.2612,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.1,23.8832,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.1,23.8832,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.1,23.8832,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.8,19.9313,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.8,19.9313,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",14.8,19.9313,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.6,31.6151,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.6,31.6151,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21.6,31.6151,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.9,28.6942,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omniscienceaccuracy-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omniscienceaccuracy-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omniscienceaccuracy-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17,23.7113,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omniscienceaccuracy-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.7,54.1237,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omniscienceaccuracy-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.7,54.1237,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omniscienceaccuracy-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",34.7,54.1237,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omniscienceaccuracy-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.4,60.4811,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omniscienceaccuracy-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.4,60.4811,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omniscienceaccuracy-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",38.4,60.4811,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omniscienceaccuracy-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.2,17.1821,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omniscienceaccuracy-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.2,17.1821,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omniscienceaccuracy-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",13.2,17.1821,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.7,59.2784,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.7,59.2784,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",37.7,59.2784,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omniscienceaccuracy-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.4,36.4261,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omniscienceaccuracy-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.4,36.4261,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omniscienceaccuracy-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.4,36.4261,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21,30.5842,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21,30.5842,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",21,30.5842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omniscienceaccuracy-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omniscienceaccuracy-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omniscienceaccuracy-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",31.4,48.4536,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.7,36.9416,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.7,36.9416,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",24.7,36.9416,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.5,29.7251,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.5,29.7251,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",20.5,29.7251,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.2,27.4914,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.2,27.4914,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",19.2,27.4914,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.2,39.5189,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.2,39.5189,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",26.2,39.5189,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.9,26.9759,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.9,26.9759,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",18.9,26.9759,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.1,46.2199,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.1,46.2199,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",30.1,46.2199,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.2,32.646,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.2,32.646,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.2,32.646,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.6,24.7423,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.6,24.7423,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",17.6,24.7423,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.7,16.323,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.7,16.323,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",12.7,16.323,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.6,21.3058,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.6,21.3058,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",15.6,21.3058,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",25.4,38.1443,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omniscienceaccuracy","Artificial Analysis Omniscience Accuracy","knowledge","Artificial Analysis","2026",22.8,33.677,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.2,22.6779,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.2,22.6779,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.2,22.6779,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",40.8,67.7925,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",40.8,67.7925,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",40.8,67.7925,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",54.9,50.7841,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",54.9,50.7841,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",54.9,50.7841,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.4,26.0555,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.4,26.0555,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.4,26.0555,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",59.8,44.8733,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",59.8,44.8733,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",59.8,44.8733,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omnisciencehallucinationrate-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",61.3,43.0639,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omnisciencehallucinationrate-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",61.3,43.0639,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-omnisciencehallucinationrate-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",61.3,43.0639,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",76,25.3317,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",76,25.3317,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",76,25.3317,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omnisciencehallucinationrate-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",36.2,73.3414,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omnisciencehallucinationrate-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",36.2,73.3414,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-omnisciencehallucinationrate-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",36.2,73.3414,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.9,54.4029,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.9,54.4029,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.9,54.4029,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",35.9,73.7033,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",35.9,73.7033,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",35.9,73.7033,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-omnisciencehallucinationrate-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",50.1,56.5742,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-omnisciencehallucinationrate-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",50.1,56.5742,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",65.9,37.5151,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",65.9,37.5151,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",65.9,37.5151,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.3,72.0145,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.3,72.0145,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.3,72.0145,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",14.1,100,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",14.1,100,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",14.1,100,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omnisciencehallucinationrate-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omnisciencehallucinationrate-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-omnisciencehallucinationrate-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.5,16.2847,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.5,16.2847,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.5,16.2847,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-omnisciencehallucinationrate-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.7,8.8058,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-omnisciencehallucinationrate-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81,19.3004,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81,19.3004,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81,19.3004,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.3,4.4632,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.3,4.4632,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.3,4.4632,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.4,11.5802,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.4,11.5802,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.4,11.5802,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.2,8.2027,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.2,8.2027,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.2,8.2027,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.9,7.3583,"percent","lower","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.9,7.3583,"percent","lower","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.9,7.3583,"percent","lower","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.9,56.8154,"percent","lower","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.9,56.8154,"percent","lower","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.9,56.8154,"percent","lower","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.7,43.7877,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.7,43.7877,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.7,43.7877,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",33.5,76.5983,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",33.5,76.5983,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",33.5,76.5983,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.5,9.047,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.5,9.047,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.5,9.047,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.8,19.5416,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.8,19.5416,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.8,19.5416,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.9,19.421,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.9,19.421,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.9,19.421,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.6,18.5766,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32.9,77.3221,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32.9,77.3221,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32.9,77.3221,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",31.3,79.2521,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",31.3,79.2521,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",31.3,79.2521,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",92.3,5.6695,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",92.3,5.6695,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",92.3,5.6695,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.1,37.2738,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.1,37.2738,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.1,37.2738,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.3,8.082,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.3,8.082,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.3,8.082,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omnisciencehallucinationrate-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34,75.9952,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omnisciencehallucinationrate-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34,75.9952,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-omnisciencehallucinationrate-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34,75.9952,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",62.2,41.9783,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",62.2,41.9783,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",62.2,41.9783,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.4,81.544,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.4,81.544,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.4,81.544,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.1,83.1122,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.1,83.1122,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.1,83.1122,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.9,35.1025,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.9,35.1025,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.9,35.1025,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.6,20.9891,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.6,20.9891,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.6,20.9891,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.4,20.0241,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.4,20.0241,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.4,20.0241,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.9,71.2907,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.9,71.2907,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",37.9,71.2907,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.1,17.9735,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.1,20.386,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.3,55.1267,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.3,55.1267,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51.3,55.1267,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.4,27.2618,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.8,29.1918,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.8,29.1918,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.8,29.1918,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.9,12.1834,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.9,12.1834,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.9,12.1834,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.6,10.1327,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.8,8.6852,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.8,8.6852,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.8,8.6852,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.6,28.2268,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.6,28.2268,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.6,28.2268,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.1,8.3233,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.1,8.3233,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",90.1,8.3233,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.8,9.8914,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.8,9.8914,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",88.8,9.8914,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.2,14.234,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.2,14.234,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.2,14.234,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.2,6.9964,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.2,6.9964,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.2,6.9964,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.1,3.4982,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.1,3.4982,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.1,3.4982,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.8,23.1604,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.8,23.1604,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.8,23.1604,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.4,16.4053,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.4,16.4053,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.4,16.4053,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.4,3.1363,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.4,3.1363,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94.4,3.1363,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omnisciencehallucinationrate-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.2,39.5657,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omnisciencehallucinationrate-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.2,39.5657,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-omnisciencehallucinationrate-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.2,39.5657,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66,37.3945,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66,37.3945,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66,37.3945,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.8,18.3353,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.8,18.3353,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.8,18.3353,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.4,29.6743,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.4,29.6743,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",72.4,29.6743,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25,86.8516,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25,86.8516,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25,86.8516,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",53.5,52.4729,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.5,22.316,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.5,22.316,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.5,22.316,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omnisciencehallucinationrate-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omnisciencehallucinationrate-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-omnisciencehallucinationrate-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73,28.9505,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omnisciencehallucinationrate-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",63.1,40.8926,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omnisciencehallucinationrate-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",63.1,40.8926,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-omnisciencehallucinationrate-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",63.1,40.8926,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.2,27.503,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.2,27.503,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",74.2,27.503,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omnisciencehallucinationrate-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omnisciencehallucinationrate-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-omnisciencehallucinationrate-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",64.6,39.0832,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",39.3,69.6019,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",39.3,69.6019,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",39.3,69.6019,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.3,20.1448,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",50.9,55.6092,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",50.9,55.6092,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",50.9,55.6092,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",47,60.3136,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",47,60.3136,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",47,60.3136,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",94,3.6188,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",95.8,1.4475,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51,55.4885,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51,55.4885,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",51,55.4885,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.3,11.7008,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.3,11.7008,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.3,11.7008,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.3,22.5573,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.3,22.5573,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",78.3,22.5573,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.1,26.4174,"percent","lower","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.1,26.4174,"percent","lower","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",75.1,26.4174,"percent","lower","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.4,63.4499,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.4,63.4499,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.4,63.4499,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.9,80.9409,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.9,80.9409,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",29.9,80.9409,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",24.5,87.4548,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",24.5,87.4548,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",24.5,87.4548,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34.4,75.5127,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34.4,75.5127,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",34.4,75.5127,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",16.1,97.5875,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",16.1,97.5875,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",16.1,97.5875,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.8,35.2232,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.8,35.2232,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",67.8,35.2232,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.7,16.0434,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.7,16.0434,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.7,16.0434,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.9,43.5464,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.9,43.5464,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",60.9,43.5464,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82,18.0941,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omnisciencehallucinationrate-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omnisciencehallucinationrate-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-omnisciencehallucinationrate-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",66.8,36.4294,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.2,28.7093,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.2,28.7093,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",73.2,28.7093,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",38.1,71.0495,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",38.1,71.0495,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",38.1,71.0495,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.9,17.0084,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.9,17.0084,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",82.9,17.0084,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.1,16.7672,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.1,16.7672,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",83.1,16.7672,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.5,82.6297,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.5,82.6297,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",28.5,82.6297,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.7,18.456,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.7,18.456,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",81.7,18.456,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.9,23.0398,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.9,23.0398,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",77.9,23.0398,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omnisciencehallucinationrate-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",69.3,33.4138,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omnisciencehallucinationrate-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",69.3,33.4138,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-omnisciencehallucinationrate-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",69.3,33.4138,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omnisciencehallucinationrate-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.1,11.9421,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omnisciencehallucinationrate-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.1,11.9421,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-omnisciencehallucinationrate-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",87.1,11.9421,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omnisciencehallucinationrate-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.5,19.9035,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omnisciencehallucinationrate-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.5,19.9035,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-omnisciencehallucinationrate-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",80.5,19.9035,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.2,63.6912,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.2,63.6912,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",44.2,63.6912,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.4,9.1677,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",79.7,20.8685,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omnisciencehallucinationrate-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omnisciencehallucinationrate-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-omnisciencehallucinationrate-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",89.1,9.5296,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",85.5,13.8721,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84,15.6815,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",48.3,58.7455,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",48.3,58.7455,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",48.3,58.7455,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32,78.4077,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32,78.4077,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",32,78.4077,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.7,57.0567,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.7,57.0567,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",49.7,57.0567,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",22.9,89.3848,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",22.9,89.3848,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",22.9,89.3848,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25.5,86.2485,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25.5,86.2485,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",25.5,86.2485,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",93.5,4.222,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",97,0,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",97,0,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",97,0,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",91.5,6.6345,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84.4,15.199,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84.4,15.199,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",84.4,15.199,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-omnisciencehallucinationrate","Artificial Analysis Omniscience Hallucination Rate","knowledge","Artificial Analysis","2026",86.6,12.5452,"percent","lower","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaopennessindex-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaopennessindex-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaopennessindex-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",50,33.4,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaopennessindex-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaopennessindex-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaopennessindex-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaopennessindex-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",44.4,22.2,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaopennessindex-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",44.4,22.2,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaopennessindex-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",44.4,22.2,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaopennessindex-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaopennessindex-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaopennessindex-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaopennessindex-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaopennessindex-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",38.9,11.2,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaopennessindex-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaopennessindex-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaopennessindex-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",33.3,0,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",83.3,100,"index","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",83.3,100,"index","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-aaopennessindex","Artificial Analysis Openness Index","knowledge","Artificial Analysis","2026",83.3,100,"index","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-protocolsunderstanding-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-protocolsunderstanding","Benchling Molecular Biology Protocols Understanding","knowledge","Benchling and Anthropic","2026",78.4,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-protocolsunderstanding-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-protocolsunderstanding","Benchling Molecular Biology Protocols Understanding","knowledge","Benchling and Anthropic","2026",78.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-biomysterybenchhumandifficult-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","2026",49.4,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-biomysterybenchhumandifficult-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","2026",49.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-biomystery-human-difficult-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","August 2026 human-difficult",34.1,34.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-biomystery-human-difficult-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","August 2026 human-difficult",41.2,41.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-biomystery-human-difficult-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","August 2026 human-difficult",43.5,43.5,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-biomystery-human-difficult-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-biomysterybenchhumandifficult","BioMysteryBench Human Difficult","knowledge","Anthropic","August 2026 human-difficult",49.4,49.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-claude-opus-5-biomysterybenchhumansolvable-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","2026",90.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-biomysterybenchhumansolvable-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","2026",90.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-biomystery-human-solvable-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","August 2026 human-solvable",87.5,87.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-biomystery-human-solvable-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","August 2026 human-solvable",80.6,80.6,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-biomystery-human-solvable-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","August 2026 human-solvable",87.1,87.1,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-biomystery-human-solvable-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-biomysterybenchhumansolvable","BioMysteryBench Human Solvable","knowledge","Anthropic","August 2026 human-solvable",83.8,83.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-claude-opus-4-5-ceval-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.2,66.6667,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ceval-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.2,66.6667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-ceval-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.2,66.6667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-ceval-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.1,63.6364,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-ceval-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.1,63.6364,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-ceval-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",92.1,63.6364,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-ceval-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.1,93.9394,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-ceval-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.1,93.9394,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-ceval-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.1,93.9394,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ceval-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93,90.9091,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ceval-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93,90.9091,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-ceval-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93,90.9091,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-ceval-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",91.4,42.4242,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-ceval-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",91.4,42.4242,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-ceval-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",91.4,42.4242,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ceval-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ceval-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-ceval-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",93.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ceval-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",90,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ceval-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",90,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ceval-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ceval","C-Eval","knowledge","C-Eval authors","2023",90,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmmlu","Chinese Massive Multitask Language Understanding","knowledge","DeepSeek-AI","2026",90.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-chinesesimpleqa-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",73.2,13.1783,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",71.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",71.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",71.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",78.9,57.3643,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-chinesesimpleqa-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",77.7,48.062,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",75.8,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",75.8,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",75.8,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-chinesesimpleqa","Chinese-SimpleQA","knowledge","DeepSeek-AI","2026",84.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-solar-open2-250b-benchlm-click-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","benchlm-click","Cultural and Linguistic Intelligence in Korean","knowledge","BenchLM registry","standard",90.7,90.7,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official Upstage Korean table owner score."],["benchlm-ref-gpt-5-6-luna-exploitbench-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",33.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-exploitbench-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",33.2,2.8916,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-exploitbench-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",33.2,2.8916,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitbench-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",73.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitbench-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",73.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-exploitbench-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",73.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitbench-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",52.9,48.8834,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitbench-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",52.9,50.3614,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-exploitbench-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",52.9,50.3614,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-exploitbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",32,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-exploitbench-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-exploitbench","ExploitBench v8-bench","knowledge","Seunghyun Lee, David Brumley, Carnegie Mellon University","2026",32,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",33.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",33.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",33.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",62.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",62.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-factsparametric","FACTS Parametric","knowledge","DeepSeek-AI","2026",62.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscience-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscience","FrontierScience","knowledge","OpenAI","2026",36.7,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscience-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscience","FrontierScience","knowledge","OpenAI","2026",36.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscience-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscience","FrontierScience","knowledge","OpenAI","2026",36.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscienceresearch-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscienceresearch","FrontierScience Research","knowledge","Meta AI","2026",36.7,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscienceresearch-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscienceresearch","FrontierScience Research","knowledge","Meta AI","2026",36.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontierscienceresearch-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontierscienceresearch","FrontierScience Research","knowledge","Meta AI","2026",36.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gmmlu-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gmmlu","Global MMLU","knowledge","Singh et al.","2024",92.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gmmlu-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gmmlu","Global MMLU","knowledge","Singh et al.","2024",92.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-gpqa-2026-07-21","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.4,48.4948,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-gpqa-2026-07-27","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.4,48.4948,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-gpqa-2026-08-01","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.4,48.4948,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-gpqa-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.1,98.0026,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-gpqa-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.1,98.0026,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-gpqa-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.1,98.0026,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gpqa-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gpqa-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-gpqa-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gpqa-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.3,94.0077,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gpqa-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.3,94.0077,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gpqa-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.3,94.0077,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqa-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.2,98.1452,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqa-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.2,98.1452,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqa-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.2,98.1452,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqa-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqa-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqa-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gpqa-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.4,82.7365,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gpqa-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.4,82.7365,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-gpqa-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.4,82.7365,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gpqa-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gpqa-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-gpqa-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gpqa-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.1,48.0668,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gpqa-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.1,48.0668,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-gpqa-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59.1,48.0668,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqa-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.4,88.4434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71.2,65.3303,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71.2,65.3303,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71.2,65.3303,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqa-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.1,89.4421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqa-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.1,90.8689,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.9,67.7557,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.9,67.7557,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.9,67.7557,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqa-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gpqa-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83,82.1658,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gpqa-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83,82.1658,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-gpqa-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83,82.1658,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqa-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.2,95.2918,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqa-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.2,95.2918,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqa-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.2,95.2918,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqa-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",78.8,76.1735,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqa-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",78.8,76.1735,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqa-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",78.8,76.1735,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gpqa-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.3,84.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gpqa-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.3,84.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-gpqa-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.3,84.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gpqa-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gpqa-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-gpqa-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gpqa-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",58.6,47.3534,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gpqa-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",58.6,47.3534,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-gpqa-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",58.6,47.3534,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gpqa-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.7,86.018,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gpqa-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.7,86.018,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-gpqa-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.7,86.018,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqa-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqa-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqa-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqa-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.2,93.865,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqa-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.2,93.865,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqa-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",91.2,93.865,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gpqa-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",66.3,58.3393,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gpqa-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",66.3,58.3393,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-gpqa-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",66.3,58.3393,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gpqa-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",64.2,55.3431,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gpqa-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",64.2,55.3431,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-gpqa-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",64.2,55.3431,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gpqa-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",50.3,35.5115,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gpqa-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",50.3,35.5115,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-gpqa-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",50.3,35.5115,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gpqa-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gpqa-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-gpqa-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.8,96.1478,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.8,96.1478,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.8,96.1478,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gpqa-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88,89.2995,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gpqa-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88,89.2995,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-gpqa-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88,89.2995,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gpqa-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",82.8,81.8804,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gpqa-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",82.8,81.8804,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-gpqa-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",82.8,81.8804,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqa-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqa-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqa-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.6,97.2892,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqa-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.3,95.4344,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqa-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.3,95.4344,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqa-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.3,95.4344,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqa-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.6,98.7159,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqa-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.6,98.7159,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqa-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",94.6,98.7159,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqa-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.9,96.2905,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqa-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.9,96.2905,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqa-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.9,96.2905,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gpqa-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gpqa-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-gpqa-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.1,92.2956,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqa-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.2,88.1581,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqa-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.2,88.1581,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqa-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.2,88.1581,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqa-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.9,89.1568,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqa-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.9,89.1568,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqa-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.9,89.1568,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-gpqa-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.5,91.4396,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqa-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqa-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqa-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",89.9,92.0103,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gpqa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gpqa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-gpqa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.6,88.7288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqa-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.5,92.8663,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqa-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.5,92.8663,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqa-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.5,92.8663,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqa-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.5,97.1465,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqa-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.5,97.1465,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqa-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",93.5,97.1465,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqa-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.41,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqa-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.41,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqa-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.41,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-gpqa-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.66,0.3567,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-gpqa-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.66,0.3567,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-gpqa-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",25.66,0.3567,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gpqa-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59,47.9241,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gpqa-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59,47.9241,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-gpqa-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",59,47.9241,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqa-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqa-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqa-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-07-21","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",40.9,22.1002,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-07-27","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",40.9,22.1002,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",40.9,22.1002,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-07-21","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.6,45.9267,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-07-27","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.6,45.9267,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.6,45.9267,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gpqa-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.7,83.1645,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gpqa-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.7,83.1645,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-gpqa-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",83.7,83.1645,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.2,66.757,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.2,66.757,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",72.2,66.757,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqa-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqa-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqa-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87,87.8727,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-gpqa-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",75.7,71.7506,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-gpqa-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",75.7,71.7506,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-gpqa-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",75.7,71.7506,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-pro-gpqa-2026-07-21","o1-pro","o1-pro","Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",79,76.4588,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-pro-gpqa-2026-07-27","o1-pro","o1-pro","Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",79,76.4588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-pro-gpqa-2026-08-01","o1-pro","o1-pro","Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",79,76.4588,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-gpqa-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.2,73.8907,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-gpqa-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.2,73.8907,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-gpqa-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.2,73.8907,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-gpqa-2026-07-21","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.5,74.3187,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-gpqa-2026-07-27","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.5,74.3187,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-gpqa-2026-08-01","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",77.5,74.3187,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gpqa-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.5,85.7326,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gpqa-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.5,85.7326,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-gpqa-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",85.5,85.7326,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gpqa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.4,89.8702,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gpqa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.4,89.8702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-gpqa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",88.4,89.8702,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86.6,87.302,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86.6,87.302,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86.6,87.302,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",84.2,83.8779,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gpqa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.8,89.0141,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gpqa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.8,89.0141,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-gpqa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",87.8,89.0141,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gpqa-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.4,92.7236,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gpqa-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.4,92.7236,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-gpqa-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.4,92.7236,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",86,86.446,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqa-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqa-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqa-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",92.4,95.5771,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.3,92.581,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.3,92.581,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",90.3,92.581,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqa-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqa-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqa-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqa-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqa-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqa-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",95.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqa-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqa-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqa-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",43.4,25.667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqa-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.3,45.4986,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqa-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.3,45.4986,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqa-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",57.3,45.4986,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqa-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71,65.0449,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqa-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71,65.0449,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqa-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-gpqa","Graduate-Level Google-Proof Q&A","knowledge","David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","2023",71,65.0449,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-solar-open2-250b-benchlm-hrm8k-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","benchlm-hrm8k","HAE-RAE Math 8K","knowledge","BenchLM registry","standard",92.2,92.2,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official Upstage Korean table owner score."],["benchlm-ref-claude-opus-5-healthbenchlengthadjusted-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchlengthadjusted","HealthBench length-adjusted score","knowledge","Anthropic","2026",57.8,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbenchlengthadjusted-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchlengthadjusted","HealthBench length-adjusted score","knowledge","Anthropic","2026",57.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbenchprofessional-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.8,94.3548,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbenchprofessional-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.8,94.3548,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2067","claude-opus-5","Claude Opus 5","HealthBench Professional as published in the Anthropic Claude Opus 5 launch table.",null,"HealthBench Professional as published in the Anthropic Claude Opus 5 launch table.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.8,59.8,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 59.8% on HealthBench Professional."],["benchlm-ref-gpt-5-4-healthbenchprofessional-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",48.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-healthbenchprofessional-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",48.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-healthbenchprofessional-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",48.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",55.7,61.2903,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",55.7,61.2903,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",55.7,61.2903,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",60.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",60.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",60.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",57.7,77.4194,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",57.7,77.4194,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",57.7,77.4194,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.3,90.3226,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.3,90.3226,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessional","HealthBench Professional","knowledge","Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","2026",59.3,90.3226,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbenchprofessionalraw-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessionalraw","HealthBench Professional raw score","knowledge","Anthropic","2026",73.4,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbenchprofessionalraw-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbenchprofessionalraw","HealthBench Professional raw score","knowledge","Anthropic","2026",73.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbench","HealthBench raw score","knowledge","Anthropic","2026",67.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-healthbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-healthbench","HealthBench raw score","knowledge","Anthropic","2026",67.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hlenotools-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",59,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hlenotools-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",59,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hlenotools-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",59,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hlenotools-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40,64.684,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hlenotools-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40,64.684,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hlenotools-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40,64.684,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hlenotools-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",46.9,77.5093,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hlenotools-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",46.9,77.5093,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hlenotools-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",46.9,77.5093,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hlenotools-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",49.8,82.8996,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hlenotools-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",49.8,82.8996,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hlenotools-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",49.8,82.8996,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hlenotools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",56.3,94.9814,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hlenotools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",56.3,94.9814,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2059","claude-opus-5","Claude Opus 5","Humanity's Last Exam without tools as published in the Anthropic Claude Opus 5 launch table.",null,"Humanity's Last Exam without tools as published in the Anthropic Claude Opus 5 launch table.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",56.3,56.3,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 56.3% on HLE without tools."],["benchlm-ref-claude-sonnet-5-hlenotools-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.2,70.632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hlenotools-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.2,70.632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hlenotools-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.2,70.632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-hlenotools-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",45.4,74.7212,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-hlenotools-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",45.4,74.7212,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-hlenotools-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",45.4,74.7212,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-hlenotools-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",5.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-hlenotools-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",5.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-hlenotools-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",5.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",8.7,6.5056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",8.7,6.5056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",8.7,6.5056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hlenotools-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",19.5,26.5799,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hlenotools-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",19.5,26.5799,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hlenotools-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",19.5,26.5799,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hlenotools-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40.5,65.6134,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hlenotools-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40.5,65.6134,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hlenotools-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",40.5,65.6134,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hlenotools-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.7,69.7026,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hlenotools-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.7,69.7026,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hlenotools-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.7,69.7026,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hlenotools-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",39.8,64.3123,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hlenotools-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",39.8,64.3123,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hlenotools-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",39.8,64.3123,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hlenotools-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",28.2,42.7509,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hlenotools-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",28.2,42.7509,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hlenotools-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",28.2,42.7509,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hlenotools-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",24.3,35.5019,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hlenotools-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",24.3,35.5019,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hlenotools-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",24.3,35.5019,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hlenotools-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.1,70.4461,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hlenotools-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.1,70.4461,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hlenotools-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.1,70.4461,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hlenotools-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",41.4,67.2862,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hlenotools-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",41.4,67.2862,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hlenotools-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",41.4,67.2862,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-hlenotools-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",31.6,49.0706,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-hlenotools-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",31.6,49.0706,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-hlenotools-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",31.6,49.0706,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hlenotools-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",30,46.0967,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hlenotools-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",30,46.0967,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hlenotools-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",30,46.0967,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-hlenotools-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",31.6,49.0706,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hlenotools-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.5,71.1896,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hlenotools-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.5,71.1896,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hlenotools-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",43.5,71.1896,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hlenotools-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",34,53.5316,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hlenotools-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",34,53.5316,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hlenotools-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",34,53.5316,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hlenotools-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.8,69.8885,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hlenotools-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.8,69.8885,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hlenotools-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",42.8,69.8885,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hlenotools-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",52.2,87.3606,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hlenotools-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",52.2,87.3606,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hlenotools-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",52.2,87.3606,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlenotools-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",26.7,39.9628,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlenotools-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",26.7,39.9628,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hlenotools-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",26.7,39.9628,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["nvidia-nemotron-3-5-lightning-hle-text-only-no-tools-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",11.72,11.72,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["benchlm-ref-sakana-fugu-hlenotools-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",47.2,78.0669,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-hlenotools-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",47.2,78.0669,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-hlenotools-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",47.2,78.0669,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-hlenotools-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",50,83.2714,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-hlenotools-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",50,83.2714,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-hlenotools-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","2026",50,83.2714,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-pro-0813-release-hle-no-tools-claude-fable-5-2026-08-13","claude-fable-5","Claude Fable 5","Fable 5 (with fallback)","claude-fable-5-deepseek-0813-release-unspecified","Fable 5 (with fallback)","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",53.3,53.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-claude-opus-4-8-2026-08-13","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8","claude-opus-4-8-deepseek-0813-release-unspecified","Claude Opus 4.8","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",49.8,49.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-deepseek-v4-flash-2026-08-13","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash Preview","deepseek-v4-flash-deepseek-0813-release-unspecified","DeepSeek V4 Flash Preview","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",34.8,34.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-deepseek-v4-flash-0731-2026-08-13","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek V4 Flash 0731","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",37.8,37.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-deepseek-v4-pro-2026-08-13","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro Preview","deepseek-v4-pro-deepseek-0813-release-unspecified","DeepSeek V4 Pro Preview","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",37.7,37.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-deepseek-v4-pro-0813-2026-08-13","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813-deepseek-0813-release-unspecified","DeepSeek V4 Pro 0813","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",42.7,42.7,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct DeepSeek release-table value; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-glm-5-2-2026-08-13","glm-5-2","GLM-5.2","GLM-5.2","glm-5-2-deepseek-0813-release-unspecified","GLM-5.2","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",40.5,40.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["deepseek-v4-pro-0813-release-hle-no-tools-kimi-k3-2026-08-13","kimi-k3","Kimi K3","Kimi K3","kimi-k3-deepseek-0813-release-unspecified","Kimi K3","benchlm-hlenotools","Humanity's Last Exam without tools","knowledge","OpenAI","August 2026, no tools",43.5,43.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-deepseek-v4-pro-0813-release","refresh-deepseek-v4-pro-0813-release","DeepSeek permanent refresh source","DeepSeek","https://api-docs.deepseek.com/news/news260813","2026-08-13","2026-08-15","2026-08-13","provider-reported","Third-party comparison value published in DeepSeek's official release table; the exact reasoning setting is not stated."],["benchlm-ref-claude-opus-4-8-include-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",87.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-include-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",87.6,67.6471,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-include-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",87.6,67.6471,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-include-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",89.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-include-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",89.8,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-include-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",86.2,69.5652,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-include-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",86.2,47.0588,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-include-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",86.2,47.0588,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-include-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",83,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-include-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",83,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-include-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-include","INCLUDE","knowledge","Qwen","2026",83,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-solar-open2-250b-benchlm-kmmlupro-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","benchlm-kmmlupro","KMMLU-Pro","knowledge","BenchLM registry","standard",78.4,78.4,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official Upstage Korean table owner score."],["benchlm-ref-claude-opus-5-singlecellbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-singlecellbench","LatchBio SingleCellBench","knowledge","LatchBio and Anthropic","2026",60.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-singlecellbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-singlecellbench","LatchBio SingleCellBench","knowledge","LatchBio and Anthropic","2026",60.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-spatialbenchverified-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-spatialbenchverified","LatchBio SpatialBench Verified","knowledge","LatchBio and Anthropic","2026",72.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-spatialbenchverified-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-spatialbenchverified","LatchBio SpatialBench Verified","knowledge","LatchBio and Anthropic","2026",72.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlu-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",88.7,73.5043,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlu-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",88.7,73.5043,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlu-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",88.7,73.5043,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlu-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.1,85.4701,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlu-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.1,85.4701,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlu-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.1,85.4701,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mmlu-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.2,86.3248,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mmlu-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.2,86.3248,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mmlu-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",90.2,86.3248,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-mmlu-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.5,63.2479,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-mmlu-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.5,63.2479,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-mmlu-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.5,63.2479,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-mmlu-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",80.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-mmlu-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",80.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-mmlu-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",80.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-mmlu-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",91.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-mmlu-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",91.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-mmlu-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",91.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-mmlu-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",86.9,58.1197,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-mmlu-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",86.9,58.1197,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-mmlu-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",86.9,58.1197,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmlu-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.2,60.6838,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmlu-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.2,60.6838,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmlu-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlu","Massive Multitask Language Understanding","knowledge","Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","2020",87.2,60.6838,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-celeris-1-mmlupro-2026-07-27","celeris-1","Celeris-1","Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",75.9,80.5065,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-celeris-1-mmlupro-2026-08-01","celeris-1","Celeris-1","Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",75.9,80.5065,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmlupro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.5,99.8577,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmlupro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.5,99.8577,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmlupro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.5,99.8577,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmlupro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82,89.1861,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmlupro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82,89.1861,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmlupro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82,89.1861,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-mmlupro-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",79.2,85.202,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-mmlupro-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",79.2,85.202,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-mmlupro-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",79.2,85.202,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-mmlupro-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",75.9,80.5065,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-mmlupro-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",75.9,80.5065,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-mmlupro-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",75.9,80.5065,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mmlupro-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.4,95.4468,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mmlupro-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mmlupro-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mmlupro-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.3,69.6927,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.3,69.6927,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.3,69.6927,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mmlupro-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mmlupro-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.9,90.4667,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mmlupro-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.9,90.4667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mmlupro-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.9,90.4667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.5,97.012,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",73.5,77.0916,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",73.5,77.0916,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",73.5,77.0916,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-mmlupro-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",81.8,88.9015,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-mmlupro-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",81.8,88.9015,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-mmlupro-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",81.8,88.9015,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmlupro-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.2,82.3563,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmlupro-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.2,82.3563,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmlupro-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.2,82.3563,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.6,90.0398,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.6,90.0398,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",82.6,90.0398,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmlupro-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmlupro-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmlupro-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-mmlupro-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",60,57.8828,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-mmlupro-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",60,57.8828,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-mmlupro-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",60,57.8828,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-mmlupro-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",69.4,71.2578,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-mmlupro-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",69.4,71.2578,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-mmlupro-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",69.4,71.2578,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-mmlupro-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.3,92.4587,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-mmlupro-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.3,92.4587,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-mmlupro-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.3,92.4587,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmlupro-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.7,94.4508,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmlupro-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.7,94.4508,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmlupro-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.7,94.4508,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmlupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmlupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmlupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmlupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmlupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmlupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.1,96.4428,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-mmlupro-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",20.25,1.3233,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-mmlupro-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",20.25,1.3233,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-mmlupro-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",20.25,1.3233,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",19.32,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",19.32,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",19.32,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-mmlupro-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85,93.4548,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-mmlupro-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85,93.4548,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-mmlupro-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85,93.4548,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-mmlupro-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.9,93.3125,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-mmlupro-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.9,93.3125,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-mmlupro-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",84.9,93.3125,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmlupro-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",48.85,42.0176,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmlupro-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",48.85,42.0176,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmlupro-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",48.85,42.0176,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.3,82.4986,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.3,82.4986,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",77.3,82.4986,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmlupro-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.8,96.0159,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmlupro-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.8,96.0159,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmlupro-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.8,96.0159,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmlupro-2026-07-21","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmlupro-2026-07-27","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmlupro-2026-08-01","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",83,90.609,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmlupro-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.1,95.0199,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmlupro-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.1,95.0199,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmlupro-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.1,95.0199,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmlupro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.8,97.4388,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmlupro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.8,97.4388,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmlupro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",87.8,97.4388,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.7,95.8736,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.7,95.8736,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.7,95.8736,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.3,93.8816,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.3,93.8816,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.3,93.8816,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmlupro-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmlupro-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmlupro-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",86.2,95.1622,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmlupro-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmlupro-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmlupro-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",85.2,93.7393,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmlupro-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmlupro-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmlupro-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",89.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmlupro-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmlupro-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmlupro-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",88.5,98.4348,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",51.4,45.646,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",51.4,45.646,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",51.4,45.646,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-mmlupro-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.1,69.4081,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-mmlupro-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.1,69.4081,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-mmlupro-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",68.1,69.4081,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-mmlupro-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",74.2,78.0876,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-mmlupro-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",74.2,78.0876,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-mmlupro-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlupro","Massive Multitask Language Understanding Professional","knowledge","Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","2024",74.2,78.0876,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-maxife-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",89.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-maxife-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",89.2,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-maxife-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",89.2,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-maxife-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",88.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-maxife-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",88.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-maxife-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-maxife","MAXIFE","knowledge","Qwen","2026",88.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-simpleqa-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",28.9,16.6667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-simpleqa-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",23.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-simpleqa-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",23.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-simpleqa-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",23.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",34.1,31.6092,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",30.1,20.1149,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",30.1,20.1149,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",30.1,20.1149,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-simpleqa-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",46.2,66.3793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-simpleqa-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",45,62.931,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-simpleqa-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",45,62.931,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-simpleqa-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",45,62.931,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",57.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",55.2,92.2414,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",55.2,92.2414,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",55.2,92.2414,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-simpleqa-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",31,22.7011,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-simpleqa-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",31,22.7011,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-simpleqa-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-simpleqa","Measuring Short-Form Factuality in Large Language Models","knowledge","Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","2024",31,22.7011,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-medxpertqatext-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.1,8.9202,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-medxpertqatext-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.1,8.9202,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-medxpertqatext-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.1,8.9202,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",71.5,100,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",71.5,100,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",71.5,100,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqatext-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",59.6,44.1315,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqatext-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",59.6,44.1315,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqatext-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",59.6,44.1315,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqatext-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",50.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqatext-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",50.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqatext-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",50.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqatext-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.6,11.2676,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqatext-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.6,11.2676,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqatext-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqatext","MedXpertQA Text","knowledge","Meta AI","2026",52.6,11.2676,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-07-21","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-07-27","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-08-01","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-07-21","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-07-27","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-08-01","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-mkqa11","MKQA-11 multilingual retrieval","knowledge","Liquid AI","2026",69.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmluproarcee-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",89.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmluproarcee-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",89.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmluproarcee-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",89.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluproarcee-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",85.8,76.259,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluproarcee-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",85.8,76.259,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluproarcee-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",85.8,76.259,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluproarcee-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",87.1,85.6115,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluproarcee-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",87.1,85.6115,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluproarcee-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",87.1,85.6115,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmluproarcee-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",80.8,40.2878,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmluproarcee-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",80.8,40.2878,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-mmluproarcee-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",80.8,40.2878,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmluproarcee-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",75.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmluproarcee-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",75.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-mmluproarcee-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",75.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-mmluproarcee-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",83.4,58.9928,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-mmluproarcee-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",83.4,58.9928,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-mmluproarcee-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluproarcee","MMLU-Pro first-party comparison snapshot","knowledge","Arcee AI","2026",83.4,58.9928,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluprox-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.7,82.8947,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluprox-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.7,82.8947,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluprox-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.7,82.8947,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluprox-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83.1,48.6842,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluprox-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83.1,48.6842,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmluprox-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83.1,48.6842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluprox-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.3,38.1579,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluprox-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.3,38.1579,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmluprox-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.3,38.1579,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmluprox-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83,47.3684,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmluprox-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83,47.3684,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-mmluprox-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",83,47.3684,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmluprox-2026-07-21","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",79.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmluprox-2026-07-27","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",79.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-mmluprox-2026-08-01","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",79.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmluprox-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmluprox-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmluprox-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluprox-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluprox-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluprox-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",82.2,36.8421,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",81,21.0526,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",81,21.0526,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",81,21.0526,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluprox-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluprox-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluprox-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",84.7,69.7368,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluprox-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",87,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluprox-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",87,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluprox-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",87,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluprox-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.4,78.9474,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluprox-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.4,78.9474,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluprox-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluprox","MMLU-ProX","knowledge","MMLU-ProX authors","2025",85.4,78.9474,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluredux-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",96.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluredux-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",96.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmluredux-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",96.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",89.4,72.8711,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",89.4,72.8711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",89.4,72.8711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",90.8,78.1462,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",90.8,78.1462,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",90.8,78.1462,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-07-21","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",78.1,30.2939,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-07-27","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",78.1,30.2939,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",78.1,30.2939,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-07-21","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",86.2,60.8139,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-07-27","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",86.2,60.8139,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",86.2,60.8139,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmluredux-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",70.06,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmluredux-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",70.06,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-mmluredux-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",70.06,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluredux-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.9,93.5946,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluredux-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.9,93.5946,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmluredux-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.9,93.5946,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmluredux-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",93.5,88.3195,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmluredux-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",93.5,88.3195,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmluredux-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",93.5,88.3195,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluredux-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluredux-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmluredux-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluredux-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",95,93.9714,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluredux-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",95,93.9714,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmluredux-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",95,93.9714,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluredux-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluredux-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmluredux-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmluredux","MMLU-Redux","knowledge","Qwen","2026",94.5,92.0874,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",88.8,72,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",88.8,72,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",88.8,72,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmlu-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",83.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmlu-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",83.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmlu-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",83.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmlu-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmlu-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmlu-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmmlu-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmmlu-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mmmlu-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",90.3,92,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmlu-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",89,74.6667,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmlu-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",89,74.6667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmlu-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmlu","MMMLU","knowledge","OpenAI","2026",89,74.6667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-protocolstroubleshooting-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-protocolstroubleshooting","Molecular Biology Protocols Troubleshooting","knowledge","Anthropic","2026",61.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-protocolstroubleshooting-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-protocolstroubleshooting","Molecular Biology Protocols Troubleshooting","knowledge","Anthropic","2026",61.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-milu-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-milu","Multi-task Indic Language Understanding Benchmark","knowledge","Verma et al.","2024",92.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-milu-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-milu","Multi-task Indic Language Understanding Benchmark","knowledge","Verma et al.","2024",92.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mgsm-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",85.7,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mgsm-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",85.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mgsm-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",85.7,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mgsm-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",84.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mgsm-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",84.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mgsm-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mgsm","Multilingual Grade School Math","knowledge","Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","2022",84.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-multiloko-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",42.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-multiloko-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",42.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-multiloko-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",42.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-multiloko-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",51.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-multiloko-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",51.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-multiloko-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-multiloko","MultiLoKo","knowledge","DeepSeek-AI","2026",51.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-07-21","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",60.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-07-27","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",60.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-08-01","lfm2-5-colbert-350m","LFM2.5-ColBERT-350M","Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",60.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-07-21","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",57.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-07-27","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",57.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-08-01","lfm2-5-embedding-350m","LFM2.5-Embedding-350M","Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","benchlm-nanobeirmultilingual","NanoBEIR Multilingual Extended","knowledge","Liquid AI","2026",57.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-nova63-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56.7,40,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-nova63-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56.7,40,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-nova63-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56.7,40,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-nova63-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",55.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-nova63-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",55.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-nova63-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",55.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-nova63-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56,22.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-nova63-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56,22.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-nova63-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",56,22.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-nova63-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59.1,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-nova63-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59.1,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-nova63-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59.1,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-nova63-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",57.9,70,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-nova63-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",57.9,70,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-nova63-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",57.9,70,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nova63-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59,97.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nova63-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59,97.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-nova63-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",59,97.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nova63-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",58.8,92.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nova63-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",58.8,92.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-nova63-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-nova63","NOVA-63","knowledge","Qwen","2026",58.8,92.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-polymath-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",86.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-polymath-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",86.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-polymath-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",86.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-polymath-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",84,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-polymath-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",84,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-polymath-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-polymath","PolyMath","knowledge","Qwen","2026",84,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-proteingymhard-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-proteingymhard","ProteinGym Hard","knowledge","Anthropic","2026",47.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-proteingymhard-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-proteingymhard","ProteinGym Hard","knowledge","Anthropic","2026",47.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-supergpqa-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.6,66.0451,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-supergpqa-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.6,66.0451,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-supergpqa-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.6,66.0451,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-supergpqa-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-supergpqa-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-supergpqa-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-supergpqa-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-supergpqa-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-supergpqa-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",95,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",46.5,32.5077,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",46.5,32.5077,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",46.5,32.5077,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",53.9,42.8055,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",53.9,42.8055,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",53.9,42.8055,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-supergpqa-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66.8,60.757,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-supergpqa-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66.8,60.757,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-supergpqa-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66.8,60.757,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-supergpqa-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",69.2,64.0969,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-supergpqa-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",69.2,64.0969,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-supergpqa-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",69.2,64.0969,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-supergpqa-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",23.14,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-supergpqa-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",23.14,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-supergpqa-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",23.14,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-supergpqa-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.9,70.6374,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-supergpqa-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.9,70.6374,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-supergpqa-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.9,70.6374,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-supergpqa-2026-07-21","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",62.6,54.9123,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-supergpqa-2026-07-27","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",62.6,54.9123,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-supergpqa-2026-08-01","qwen3-235b-2507","Qwen3 235B 2507","Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",62.6,54.9123,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-supergpqa-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",65.6,59.0871,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-supergpqa-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",65.6,59.0871,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-supergpqa-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",65.6,59.0871,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-supergpqa-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.4,65.7668,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-supergpqa-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.4,65.7668,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-supergpqa-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",70.4,65.7668,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",67.1,61.1745,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",67.1,61.1745,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",67.1,61.1745,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",63.4,56.0256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",63.4,56.0256,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",63.4,56.0256,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-supergpqa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66,59.6438,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-supergpqa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66,59.6438,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-supergpqa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",66,59.6438,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-supergpqa-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.6,67.4367,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-supergpqa-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.6,67.4367,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-supergpqa-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.6,67.4367,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",64.7,57.8347,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",64.7,57.8347,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",64.7,57.8347,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-supergpqa-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.6,70.2199,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-supergpqa-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.6,70.2199,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-supergpqa-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",73.6,70.2199,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-supergpqa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.4,67.1584,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-supergpqa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.4,67.1584,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-supergpqa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-supergpqa","SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","knowledge","Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","2025",71.4,67.1584,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swemultilingual-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",92.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swemultilingual-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",92.2,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-swemultilingual-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",92.2,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swemultilingual-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.5,63.4328,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swemultilingual-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.5,63.4328,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-swemultilingual-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.5,63.4328,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-multilingual:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.5,77.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Multilingual uses mini-SWE-agent with temperature 1, top_p 0.95 and 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["benchlm-ref-claude-opus-4-8-swemultilingual-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",84.4,80.597,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swemultilingual-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",84.4,80.597,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-swemultilingual-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",84.4,80.597,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swemultilingual-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",89.5,93.2836,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-swemultilingual-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",89.5,93.2836,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultilingual-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultilingual-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-swemultilingual-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swemultilingual-2026-07-21","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.7,53.9801,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swemultilingual-2026-07-27","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.7,53.9801,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-swemultilingual-2026-08-01","composer-2","Composer 2","Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.7,53.9801,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-swemultilingual-2026-07-21","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",79.8,69.1542,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-swemultilingual-2026-07-27","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",79.8,69.1542,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-composer-2-5-swemultilingual-2026-08-01","composer-2-5","Composer 2.5","Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",79.8,69.1542,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-swemultilingual-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",70.2,45.2736,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swemultilingual-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.7,44.0299,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swemultilingual-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.7,44.0299,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-swemultilingual-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.7,44.0299,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-swemultilingual-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",74.1,54.9751,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swemultilingual-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.8,44.2786,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swemultilingual-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.8,44.2786,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-swemultilingual-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.8,44.2786,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.2,60.199,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swemultilingual-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swemultilingual-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-swemultilingual-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.3,52.9851,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swemultilingual-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78,64.6766,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swemultilingual-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78,64.6766,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-swemultilingual-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78,64.6766,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swemultilingual-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73,52.2388,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swemultilingual-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73,52.2388,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-swemultilingual-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73,52.2388,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swemultilingual-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.7,61.4428,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swemultilingual-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.7,61.4428,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-swemultilingual-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.7,61.4428,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swemultilingual-2026-07-21","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",63.1,27.6119,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swemultilingual-2026-07-27","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",63.1,27.6119,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-m-1-swemultilingual-2026-08-01","laguna-m-1","Laguna M.1","Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",63.1,27.6119,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swemultilingual-2026-07-21","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.5,65.9204,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swemultilingual-2026-07-27","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.5,65.9204,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-s-2-1-swemultilingual-2026-08-01","laguna-s-2-1","Laguna S 2.1","Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.5,65.9204,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-laguna-xs-2-1-benchlm-swemultilingual-benchmark-2-2025","laguna-xs-2-1","Laguna XS 2.1","Laguna XS 2.1 (thinking enabled)","laguna-xs-2-1-default","Laguna XS 2.1 (thinking enabled)","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",63.1,63.1,"percent","higher","2.1.0","reference-only","direct","2026-07-02","2026-07-02","production::poolside-laguna-xs-2-1-huggingface-2026-08-15","poolside-laguna-xs-2-1-huggingface-2026-08-15","Laguna XS 2.1 model card","Poolside","https://huggingface.co/poolside/Laguna-XS-2.1","2026-07-02","2026-08-15","2026-08-15","provider-reported","Official owner score; mean pass@1 over 4 attempts on Harbor/pool. Also stated in the 2026-07-02 blog."],["benchlm-ref-laguna-xs-2-swemultilingual-2026-07-21","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",57.7,14.1791,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-swemultilingual-2026-07-27","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",57.7,14.1791,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-laguna-xs-2-swemultilingual-2026-08-01","laguna-xs-2","Laguna XS.2","Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",57.7,14.1791,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-longcat-2-0-benchlm-swemultilingual-benchmark-2-2025","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.3,77.3,"percent","higher","2.1.0","reference-only","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score for SWE-bench Multilingual."],["benchlm-ref-minimax-m2-7-swemultilingual-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.5,60.9453,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swemultilingual-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.5,60.9453,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-swemultilingual-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",76.5,60.9453,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-swemultilingual-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.7,39.0547,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-swemultilingual-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.7,39.0547,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-swemultilingual-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.7,39.0547,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["nvidia-nemotron-3-5-lightning-swe-bench-multilingual-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",39.33,39.33,"percent","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["benchlm-ref-ornith-1-0-35b-swemultilingual-2026-07-21","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.3,43.0348,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-swemultilingual-2026-07-27","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.3,43.0348,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-35b-swemultilingual-2026-08-01","ornith-1-0-35b","Ornith-1.0-35B","Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",69.3,43.0348,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swemultilingual-2026-07-21","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.9,66.9154,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swemultilingual-2026-07-27","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.9,66.9154,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-397b-swemultilingual-2026-08-01","ornith-1-0-397b","Ornith-1.0-397B","Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.9,66.9154,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swemultilingual-2026-07-21","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",52,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swemultilingual-2026-07-27","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",52,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ornith-1-0-9b-swemultilingual-2026-08-01","ornith-1-0-9b","Ornith-1.0-9B","Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",52,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swemultilingual-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",71.3,48.01,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swemultilingual-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",71.3,48.01,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-swemultilingual-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",71.3,48.01,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swemultilingual-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.8,54.2289,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swemultilingual-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.8,54.2289,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-swemultilingual-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.8,54.2289,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.2,37.8109,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.2,37.8109,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",67.2,37.8109,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swemultilingual-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swemultilingual-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-swemultilingual-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",78.3,65.4229,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swemultilingual-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",75.8,59.204,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swemultilingual-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",75.8,59.204,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-swemultilingual-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",75.8,59.204,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-multilingual:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",75.8,75.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Multilingual uses mini-SWE-agent with temperature 1, top_p 0.95 and 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:swe-multilingual:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",73.8,73.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","SWE-bench Multilingual uses mini-SWE-agent with temperature 1, top_p 0.95 and 256K context. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["benchlm-ref-swe-1-7-swemultilingual-2026-07-21","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.8,64.1791,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-swemultilingual-2026-07-27","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.8,64.1791,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-swe-1-7-swemultilingual-2026-08-01","swe-1-7","SWE-1.7","Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","benchlm-swemultilingual-benchmark-2","SWE-bench Multilingual","knowledge","SWE-bench team","2025",77.8,64.1791,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lastonescyberrangesteps-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-lastonescyberrangesteps","The Last Ones Average Progress","knowledge","NIST CAISI and UK AISI","2026",17,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lastonescyberrangesteps-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-lastonescyberrangesteps","The Last Ones Average Progress","knowledge","NIST CAISI and UK AISI","2026",17,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lastonescyberrangecompletion-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-lastonescyberrangecompletion","The Last Ones Cyber Range Completion Rate","knowledge","NIST CAISI and UK AISI","2026",10,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lastonescyberrangecompletion-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-lastonescyberrangecompletion","The Last Ones Cyber Range Completion Rate","knowledge","NIST CAISI and UK AISI","2026",10,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",82.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",82.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",82.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",85.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",85.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-triviaqa","TriviaQA","knowledge","DeepSeek-AI","2026",85.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-benchlm-skillsbench-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-skillsbench","Vals SkillsBench","knowledge","Vals AI","86 tasks / with skills",44.3,44.3,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only exact with-skills result."],["evidence-2026-08-muse-glimmer-30b-benchlm-skillsbench-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-skillsbench","Vals SkillsBench","knowledge","Vals AI","86 tasks / with skills",44.3,44.3,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only exact with-skills result."],["google-gemini-37-eval-gdm-mrcr-v2-128k-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","gdm-mrcr-v2-128k-average","GDM-MRCR v2 8-needle, 128K average","long-context","Google DeepMind","v2, eight-needle, 128K average",81.5,81.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdm-mrcr-v2-128k-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","gdm-mrcr-v2-128k-average","GDM-MRCR v2 8-needle, 128K average","long-context","Google DeepMind","v2, eight-needle, 128K average",91.8,91.8,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdm-mrcr-v2-128k-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","gdm-mrcr-v2-128k-average","GDM-MRCR v2 8-needle, 128K average","long-context","Google DeepMind","v2, eight-needle, 128K average",97,97,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-gdm-mrcr-v2-128k-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","gdm-mrcr-v2-128k-average","GDM-MRCR v2 8-needle, 128K average","long-context","Google DeepMind","v2, eight-needle, 128K average",93.5,93.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-07-562","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",64.4,64.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-343","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",77.3,77.3,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-564","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",60.8,60.8,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-296","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",61,61,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-563","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",62,62,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-342","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",90.4,90.4,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-341","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","longbench-v2","LongBench v2","long-context","THUDM","2",91.7,91.7,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["benchlm-ref-agents-a1-longbenchv2-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-longbenchv2-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-longbenchv2-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-longbenchv2-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",64.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-longbenchv2-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",64.4,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-longbenchv2-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",64.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",44.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",44.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",44.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",51.5,34.5178,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",51.5,34.5178,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",51.5,34.5178,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-longbenchv2-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.8,81.7259,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-longbenchv2-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.8,81.7259,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-longbenchv2-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.8,81.7259,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-longbenchv2-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61,82.7411,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-longbenchv2-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61,82.7411,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-longbenchv2-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61,82.7411,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-longbenchv2-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61.9,87.3096,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-longbenchv2-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61.9,87.3096,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-longbenchv2-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",61.9,87.3096,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-longbenchv2-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.6,80.7107,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-longbenchv2-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.6,80.7107,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-longbenchv2-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.6,80.7107,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-longbenchv2-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",63.2,93.9086,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-longbenchv2-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",63.2,93.9086,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-longbenchv2-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",63.2,93.9086,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",60.2,78.6802,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",59,72.5888,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",59,72.5888,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",59,72.5888,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-longbenchv2-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",62,87.8173,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-longbenchv2-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",62,87.8173,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-longbenchv2-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","longbench-v2","LongBench v2","long-context","THUDM","2025",62,87.8173,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-aime2025-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",87,82.3259,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-aime2025-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",87,82.3259,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-aime2025-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",87,82.3259,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aime2025-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",85.3,79.3213,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aime2025-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",85.3,79.3213,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aime2025-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",85.3,79.3213,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aime2025-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",95.7,97.7024,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aime2025-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",95.7,97.7024,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aime2025-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",95.7,97.7024,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aime2025-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aime2025-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aime2025-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",96.1,98.4093,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",42.53,3.7292,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",42.53,3.7292,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",42.53,3.7292,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2025-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",97,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2025-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",97,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2025-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",97,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aime2025-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",94.1,94.8745,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aime2025-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",94.1,94.8745,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aime2025-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",94.1,94.8745,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-aime2025-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",40.42,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-aime2025-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",40.42,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-aime2025-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",40.42,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",82.1,73.6656,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",82.1,73.6656,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aime-2025","AIME 2025","mathematics","MAA / public contest evals","2025",82.1,73.6656,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1406","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,73.3333,73.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1396","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,80.3333,80.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-1778","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,56.3333,56.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1729","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,74.3333,74.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1383","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,88,88,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1542","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,89.6667,89.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1528","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,92,92,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1841","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,78.3333,78.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1803","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,87.6667,87.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1743","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,86,86,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1556","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,95,95,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1343","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,94.3333,94.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1343--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,94.3333,94.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1328","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,98.6667,98.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1328--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,98.6667,98.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1355","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,85,85,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1355--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,85,85,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1306","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,99,99,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1306--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,99,99,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1873","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,84.6667,84.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1662","grok-4","Grok 4","Grok 4",null,"Grok 4","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,92.6667,92.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1766","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,89.6667,89.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1895","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,43.3333,43.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1852","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,57.3333,57.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1673","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,94.6667,94.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-765","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,96.1,96.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1592","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,96.3333,96.3333,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1754","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,78.3333,78.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1693","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,82.6667,82.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1650","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,89,89,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1368","o3","o3","o3",null,"o3","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,88.3333,88.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1815","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,90.6667,90.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1815--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","aime-2025","AIME 2025","mathematics","MAA / public contest evals",null,90.6667,90.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["benchlm-ref-claude-opus-4-5-aime2026-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.1,93.0248,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aime2026-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.1,93.0248,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aime2026-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.1,93.0248,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-command-a-plus-aime-2026-2026-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",96,96,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-aime-2026-2026-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",97,97,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-gemma-4-12b-aime2026-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",77.5,63.0827,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aime2026-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",77.5,63.0827,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aime2026-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",77.5,63.0827,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2026-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2026-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2026-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aime2026-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aime2026-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aime2026-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aime2026-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",99.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aime2026-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",99.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aime2026-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",99.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aime2026-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",97.1,96.4274,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aime2026-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",97.1,96.4274,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aime2026-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",97.1,96.4274,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-aime2026-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.5,93.7053,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2026-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2026-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2026-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.8,94.2157,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aime2026-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",96.4,95.2365,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aime2026-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",96.4,95.2365,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aime2026-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",96.4,95.2365,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",50,16.2981,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",50,16.2981,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",50,16.2981,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2026-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.5,92.0041,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2026-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.5,92.0041,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-aime2026-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.5,92.0041,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-mimo-v2-5-aime-2026-2026-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",92.3,92.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-minicpm5-1b-aime2026-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",40.42,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-aime2026-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",40.42,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-aime2026-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",40.42,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-mistral-medium-3-5-aime-2026-2026-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",89,89,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-qwen3-5-397b-aime2026-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",93.3,89.9626,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aime2026-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",93.3,89.9626,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aime2026-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",93.3,89.9626,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aime2026-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.1,91.3236,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aime2026-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.1,91.3236,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aime2026-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",94.1,91.3236,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aime2026-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aime2026-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aime2026-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.3,93.3651,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",92.7,88.9418,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",92.7,88.9418,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",92.7,88.9418,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-solar-open-100b-reasoning-aime-2026-2026-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",87.7,87.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-aime-2026-2026","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",95.7,95.7,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-zaya1-74b-preview-aime2026-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",76.4,61.2113,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-aime2026-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",76.4,61.2113,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-aime2026-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",76.4,61.2113,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-aime2026-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",89.1,82.8173,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-aime2026-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",89.1,82.8173,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-aime2026-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026",89.1,82.8173,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-aime-2026-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026 / 30 questions",94.7,94.7,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported accuracy over the complete 2026 AIME set."],["evidence-2026-08-muse-glimmer-30b-aime-2026-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","aime-2026","AIME 2026","mathematics","MAA / public contest evals","2026 / 30 questions",94.7,94.7,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported accuracy over the complete 2026 AIME set."],["evidence-2026-07-771","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,95.1,95.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-767","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,95.8,95.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-769","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,95.3,95.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-766","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,99.2,99.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-768","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,95.8,95.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-770","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","aime-2026","AIME 2026","mathematics","MAA / public contest evals",null,95.3,95.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-claude-opus-4-6-aime2025arcee-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",99.8,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aime2025arcee-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",99.8,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aime2025arcee-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",99.8,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2025arcee-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",93.3,91.4248,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2025arcee-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",93.3,91.4248,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aime2025arcee-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",93.3,91.4248,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025arcee-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025arcee-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aime2025arcee-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aime2025arcee-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",80,73.8786,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aime2025arcee-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",80,73.8786,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aime2025arcee-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",80,73.8786,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aime2025arcee-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",24,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aime2025arcee-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",24,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aime2025arcee-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",24,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aime2025arcee-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aime2025arcee-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aime2025arcee-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2025arcee","AIME25 first-party comparison snapshot","mathematics","Arcee AI","2026",96.3,95.3826,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aime2024-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2024","American Invitational Mathematics Examination 2024","mathematics","Mathematical Association of America","2024",87.3,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aime2024-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2024","American Invitational Mathematics Examination 2024","mathematics","Mathematical Association of America","2024",87.3,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aime2024-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","benchlm-aime2024","American Invitational Mathematics Examination 2024","mathematics","Mathematical Association of America","2024",87.3,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apex-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",19.1,42.4036,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apex-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",1,1.3605,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apex-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",1,1.3605,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apex-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",1,1.3605,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apex-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",33,73.9229,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apex-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",27.4,61.2245,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apex-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",0.4,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apex-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",0.4,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apex-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",0.4,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apex-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",38.3,85.941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-apex-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",44.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-apex-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",44.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-apex-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",44.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apex-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",22.7,50.5669,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apex-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",22.7,50.5669,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-apex-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",22.7,50.5669,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-apex-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",32.2,72.1088,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-apex-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",32.2,72.1088,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-apex-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-apex","Apex","mathematics","DeepSeek-AI","2026",32.2,72.1088,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-apexshortlist-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",72.1,77.6543,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apexshortlist-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.3,0.1235,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apexshortlist-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.3,0.1235,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-apexshortlist-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.3,0.1235,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.7,94.4444,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-apexshortlist-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",85.5,94.1975,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apexshortlist-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.2,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apexshortlist-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.2,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-apexshortlist-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",9.2,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-apexshortlist","Apex Shortlist","mathematics","DeepSeek-AI","2026",90.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaaime2025-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",99,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaaime2025-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",99,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaaime2025-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",99,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaaime2025-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",93.4,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaaime2025-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",93.4,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaaime2025-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-aaaime2025","Artificial Analysis AIME 2025","mathematics","Artificial Analysis","2026",93.4,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aamath500-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.4,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aamath500-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aamath500-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.2,33.3333,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aamath500-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.2,33.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aamath500-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-aamath500","Artificial Analysis MATH-500","mathematics","Artificial Analysis","2026",99.2,33.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arxivmathjune2026withtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arxivmathjune2026withtools","ArXivMath June 2026 with tools","mathematics","MathArena and Anthropic","2026",91.3,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arxivmathjune2026withtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arxivmathjune2026withtools","ArXivMath June 2026 with tools","mathematics","MathArena and Anthropic","2026",91.3,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arxivmathjune2026-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arxivmathjune2026","ArXivMath June 2026 without tools","mathematics","MathArena and Anthropic","2026",90.8,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arxivmathjune2026-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arxivmathjune2026","ArXivMath June 2026 without tools","mathematics","MathArena and Anthropic","2026",90.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-397","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",83.1,83.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-392","claude-mythos-5","Claude Mythos 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",82,82,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-525","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",58.7,58.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-458","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",54,54,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-439","claude-opus-4-7","Claude Opus 4.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",65.7,65.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-407","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",73.6,73.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-513","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",49.1,49.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-556","deepseek-v4-pro","DeepSeek V4 Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",24.8,24.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-552","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",50.1,50.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-471","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",59.7,59.7,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-425","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",61.7,61.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-493","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",51.5,51.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-463","glm-5-1","GLM-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",64.4,64.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-533","glm-5-2","GLM-5.2","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",81.2,81.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-519","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",47.4,47.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-431","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",70.2,70.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-482","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",49,49,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-419","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",70.3,70.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-537","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","gpt-5-5-pro-default-high","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",71,71,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-476","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",95.3,95.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-402","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",95.3,95.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-436","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",95.3,95.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-529","minimax-m2-7","MiniMax M2.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",63.5,63.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-452","minimax-m3","MiniMax M3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",68.8,68.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-445","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",55.5,55.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-507","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-mathematics","BenchLM Math prior","mathematics","BenchLM","bench-align-v5.1",62.9,62.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (math) used to estimate missing category coverage under methodology 1.3.0."],["benchlm-ref-deepseek-v4-flash-base-cmath-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",93.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cmath-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",93.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cmath-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",93.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmath-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",90.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmath-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",90.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cmath-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cmath","CMath","mathematics","DeepSeek-AI","2026",90.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-frontiermath-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",43.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-frontiermath-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",43.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-frontiermath-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",43.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermath-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",50,13.7168,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermath-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",50,13.7168,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermath-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",50,13.7168,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermath-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",52.4,19.0265,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermath-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",52.4,19.0265,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermath-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",52.4,19.0265,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermath-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",51.7,17.4779,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermath-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",51.7,17.4779,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermath-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",51.7,17.4779,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermath-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",78.6,76.9912,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermath-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",78.6,76.9912,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermath-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",78.6,76.9912,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermath-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",89,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermath-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",89,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermath-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",89,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermath-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",84.9,90.9292,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermath-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",84.9,90.9292,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermath-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","frontiermath","FrontierMath","mathematics","Epoch AI","2024",84.9,90.9292,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-311","claude-haiku-4-5","Claude Haiku 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,5.903,5.903,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-672","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,20.69,20.69,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-663","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,40.7,40.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-662","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,43.793,43.793,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-307","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,47.241,47.241,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-668","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,32.4,32.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-666","gemini-3-flash","Gemini 3 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,35.64,35.64,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-665","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,37.6,37.6,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-309","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,36.9,36.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-308","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,38.966,38.966,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-673","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,16.434,16.434,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-667","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,33.448,33.448,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-669","gpt-5-1","GPT-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,31.034,31.034,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-306","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,47.6,47.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-671","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,25.86,25.86,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-263","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,51.7,51.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-661","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpt-5-5-pro-default-high","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,52.4,52.4,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-262","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,78.6,78.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-260","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,89,89,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-261","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,84.9,84.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-310","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,27.9,27.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-664","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,39,39,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-670","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","frontiermath","FrontierMath","mathematics","Epoch AI",null,26.207,26.207,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-07-21","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-07-27","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-08-01","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-07-21","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-07-27","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.9,27.5904,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.9,27.5904,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.9,27.5904,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.917,27.6108,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.917,27.6108,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",22.917,27.6108,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",31.25,37.6506,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",31.25,37.6506,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",31.25,37.6506,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.3,10,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.3,10,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.3,10,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.75,22.5904,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.75,22.5904,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.75,22.5904,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",16.7,20.1205,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",16.7,20.1205,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",16.7,20.1205,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.583,17.5699,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.583,17.5699,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.583,17.5699,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tier4-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.128,2.5639,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tier4-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.128,2.5639,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tier4-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.128,2.5639,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tier4-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tier4-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tier4-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tier4-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tier4-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tier4-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.1,2.5301,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tier4-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tier4-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tier4-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",12.5,15.0602,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.8,22.6506,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.8,22.6506,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",18.8,22.6506,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tier4-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",37.5,45.1807,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tier4-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",37.5,45.1807,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tier4-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",37.5,45.1807,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",27.1,32.6506,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",27.1,32.6506,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",27.1,32.6506,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.08,2.506,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.08,2.506,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.08,2.506,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tier4-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",39.6,47.7108,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tier4-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",39.6,47.7108,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tier4-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",39.6,47.7108,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",35.4,42.6506,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",35.4,42.6506,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",35.4,42.6506,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",58.5,70.4819,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",58.5,70.4819,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",58.5,70.4819,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",83,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",83,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",83,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",68.3,82.2892,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",68.3,82.2892,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",68.3,82.2892,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-07-21","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-07-27","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-08-01","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tier4-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tier4-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tier4-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tier4-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tier4-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tier4-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.2,5.0602,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.2,5.0602,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.2,5.0602,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.58,17.5663,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.58,17.5663,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.58,17.5663,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tier4-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.6,17.5904,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tier4-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.6,17.5904,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tier4-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",14.6,17.5904,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tier4-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tier4-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tier4-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-21--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-27--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-08-01--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-21","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-27","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-08-01","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",6.25,7.5301,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",4.167,5.0205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-07-21","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-07-27","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-08-01","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-07-21","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-07-27","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-08-01","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-07-21","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-07-27","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-08-01","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",2.083,2.5096,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.333,10.0398,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.333,10.0398,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","2026",8.333,10.0398,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["epoch-frontier4-ReDAtJGGfRtrEGGWf4Zbsd","claude-fable-5","Claude Fable 5","Claude Fable 5 (max)","claude-fable-5-max","Claude Fable 5 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",87.804878,87.804878,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-09","2026-06-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-09","2026-09-01","2026-08-17","source-checked","verification_code:0.88±0.05; stderr=0.051739520574625435."],["epoch-frontier4-9zT72xRR6wZhiwraJTNPyz","claude-opus-4-1","Claude Opus 4.1","Claude Opus 4.1 (32k thinking)","claude-opus-4-1-epoch-claude-opus-4-1-20250805-32k","Claude Opus 4.1 (32k thinking)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",2.439024,2.439024,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-18","source-checked","verification_code:0.02±0.02; stderr=0.024390243902439022."],["epoch-frontier4-azk4theXxAsizmoi2HDD49","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (32k thinking)","claude-opus-4-5-epoch-claude-opus-4-5-20251101-32k","Claude Opus 4.5 (32k thinking)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",4.878049,4.878049,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.05±0.03; stderr=0.03405912205797303."],["epoch-frontier4-2hbbp8z6PhB7c2T33os9q5","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (max)","claude-opus-4-6-max","Claude Opus 4.6 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",26.829268,26.829268,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.27±0.07; stderr=0.07005564203095158."],["epoch-frontier4-P4LXQ9PyDDziEANMB5jVRd","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (max)","claude-opus-4-7-max","Claude Opus 4.7 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",31.707317,31.707317,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.32±0.07; stderr=0.07357611282438223."],["epoch-frontier4-JiqgqBTfkGx797uowZgJbk","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max)","claude-opus-4-8-max","Claude Opus 4.8 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",56.097561,56.097561,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.56±0.08; stderr=0.07846686801046543."],["epoch-frontier4-ji3kMn49Jcr9t4whxcpjj4","claude-opus-5","Claude Opus 5","Claude Opus 5 (max)","claude-opus-5-max","Claude Opus 5 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",73.170732,73.170732,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-24","2026-07-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-24","2026-09-01","2026-08-17","source-checked","verification_code:0.73±0.07; stderr=0.07005564203095158."],["epoch-frontier4-d3m9Xip8cof4s77z6LKkEh","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (32k thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929-32k","Claude Sonnet 4.5 (32k thinking)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",2.439024,2.439024,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.02±0.02; stderr=0.024390243902439022."],["epoch-frontier4-bqiVinHXFcgyFGPstJwXJN","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max)","claude-sonnet-5-max","Claude Sonnet 5 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",29.268293,29.268293,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-30","2026-06-30","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-30","2026-09-01","2026-08-17","source-checked","verification_code:0.29±0.07; stderr=0.07194088392074452."],["epoch-frontier4-WYs83xB6NwrkMNyQJtuMcQ","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731 (max)","deepseek-v4-flash-0731-epoch-deepseek-v4-flash-0731-max","DeepSeek V4 Flash 0731 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",24.390244,24.390244,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0.24±0.07; stderr=0.06789956540036612."],["epoch-frontier4-RAZ6VMfGtQSE8EvUZRRz4Q","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek v4 (max)","deepseek-v4-pro-max","DeepSeek v4 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",2.439024,2.439024,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-17","2026-06-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-17","2026-09-01","2026-08-18","source-checked","verification_code:0.02±0.02; stderr=0.024390243902439025."],["epoch-frontier4-2z7kL9DpfvY44T9Wo99FYx","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro","gemini-2-5-pro-epoch-gemini-2-5-pro","Gemini 2.5 Pro","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",0,0,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0±0; stderr=0.0."],["epoch-frontier4-ZczEhu69FQ9Cnns8NT6ew4","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview","gemini-3-flash-epoch-gemini-3-flash-preview","Gemini 3 Flash Preview","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",17.073171,17.073171,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.17±0.06; stderr=0.05949419959829497."],["epoch-frontier4-eXVWr3RBeSsWy5DAU8ed46","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview","gemini-3-1-pro-preview-epoch-gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",26.829268,26.829268,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.27±0.07; stderr=0.07005564203095158."],["epoch-frontier4-cXn5zzns6VFwbTgotDwZjV","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",26.829268,26.829268,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.27±0.07; stderr=0.07005564203095158."],["epoch-frontier4-juNVfKBcYTwaKnk9GRA2zL","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Gemini 3.5 Flash-Lite (high)","gemini-3-5-flash-lite-livebench-2026-06-25-high","Gemini 3.5 Flash-Lite (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",0,0,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0±0; stderr=0.0."],["epoch-frontier4-JCMjXU6FRiiLV3xvs9R2Xv","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high)","gemini-3-6-flash-high","Gemini 3.6 Flash (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",21.95122,21.95122,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0.22±0.07; stderr=0.06544589202438408."],["epoch-frontier4-KptKXK5nE2Ns2WbiaSSnbc","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high)","gemini-3-7-flash-high","Gemini 3.7 Flash (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",36.585366,36.585366,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","verification_code:0.37±0.08; stderr=0.07615851217559022."],["epoch-frontier4-JebA8dYLVov6H49ZQSLXh5","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",29.268293,29.268293,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-19","2026-06-19","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-19","2026-09-01","2026-08-17","source-checked","verification_code:0.29±0.07; stderr=0.07194088392074452."],["epoch-frontier4-4S3ZrJDkwh4gmGrmJQdvzR","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",29.268293,29.268293,"percent","higher","2.2.0","reference-only","direct","2026-08-25","2026-08-25","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-25","2026-09-01","2026-09-01","independently-verified","verification_code:0.29±0.07; stderr=0.07194088392074452; Epoch run 4S3ZrJDkwh4gmGrmJQdvzR. frontiermath:v2:tier-4 is an existing reference-only protocol and cannot alter generic scoring."],["epoch-frontier4-mSS7C84hJyYi4VnJtgzYvW","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",21.95122,21.95122,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.22±0.07; stderr=0.06544589202438408."],["epoch-frontier4-2WxRbjq5ZWTgiXZPqYUG4R","gpt-5-mini","GPT-5 mini","GPT-5 mini (high)","gpt-5-mini-high","GPT-5 mini (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",12.195122,12.195122,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.12±0.05; stderr=0.051739520574625435."],["epoch-frontier4-P5v3FUnVVTbaqaCr43deKi","gpt-5-nano","GPT-5 nano","GPT-5 nano (high)","gpt-5-nano-high","GPT-5 nano (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",2.439024,2.439024,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-18","source-checked","verification_code:0.02±0.02; stderr=0.024390243902439025."],["epoch-frontier4-J3NUxsK2GXq9Whkjn7mEVm","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",31.7,31.7,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.32±0.07; stderr=0.073."],["epoch-frontier4-oALBfwNc7dcNSDJSSBJXYe","gpt-5-2","GPT-5.2","GPT-5.2 Pro (xhigh)","gpt-5-2-pro-epoch-gpt-5-2-pro-2025-12-11-xhigh","GPT-5.2 Pro (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",46,46,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-18","source-checked","verification_code:0.46±0.08; stderr=0.07789619230571372."],["epoch-frontier4-doiQDghqnLzbE8CFijLNQJ","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",49,49,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.49±0.08; stderr=0.08."],["epoch-frontier4-W6b3C8AeNC7CAFs5DQma8N","gpt-5-4","GPT-5.4","GPT-5.4 Pro (xhigh)","gpt-5-4-pro-xhigh","GPT-5.4 Pro (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",58.536585,58.536585,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-18","source-checked","verification_code:0.59±0.08; stderr=0.07789619230571372."],["epoch-frontier4-jtQ6rDiNxjNKkVMEPBE3Ub","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",9.756098,9.756098,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.1±0.05; stderr=0.04691557088212523."],["epoch-frontier4-SF6FYnnKztPCDWQc4v4EJa","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (high)","gpt-5-4-nano-epoch-gpt-5-4-nano-2026-03-17-high","GPT-5.4 nano (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",12.195122,12.195122,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.12±0.05; stderr=0.051739520574625435."],["epoch-frontier4-PKSMspJiJdNThB6qNvaBSM","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",72.5,72.5,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.73±0.07; stderr=0.07149950690165274."],["epoch-frontier4-SFKXCPUu73mwxMoUotiZM2","gpt-5-5","GPT-5.5","GPT-5.5 Pro (xhigh)","gpt-5-5-pro","GPT-5.5 Pro (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",78.04878,78.04878,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.78±0.07; stderr=0.06544589202438408."],["epoch-frontier4-NLiiJJSyA9wmg2FHENFpAq","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",60.97561,60.97561,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.61±0.08; stderr=0.07712872341874097."],["epoch-frontier4-QJ5rJMVypB4PMxPcBGftc8","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",82.926829,82.926829,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.83±0.06; stderr=0.05949419959829497."],["epoch-frontier4-cdWwJSH3jVfrKyhUHmT54T","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (pro, max)","gpt-5-6-sol-max","GPT-5.6 Sol (pro, max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",80.487805,80.487805,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.8±0.06; stderr=0.06265967111543964."],["epoch-frontier4-ZUygpkXqg3NjchXXfedAWV","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",70.731707,70.731707,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.71±0.07; stderr=0.07194088392074452."],["epoch-frontier4-SX5FCWMk3HF83gGsP3WFA6","grok-4-20","Grok 4.20","grok-4.20-0309-reasoning","grok-4-20-epoch-grok-4-20-0309-reasoning","grok-4.20-0309-reasoning","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",17.073171,17.073171,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","verification_code:0.17±0.06; stderr=0.05949419959829495."],["epoch-frontier4-3iRQkf5v7Wu7YuZhxa76Gg","grok-4-3","Grok 4.3","grok-4.3_high","grok-4-3-epoch-grok-4-3-high","grok-4.3_high","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",14.634146,14.634146,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-17","2026-06-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-17","2026-09-01","2026-08-17","source-checked","verification_code:0.15±0.06; stderr=0.055885069450680974."],["epoch-frontier4-7vThbrtbczScg2UcsAYXtF","grok-4-5","Grok 4.5","Grok 4.5 (high)","grok-4-5-aa-2-high","Grok 4.5 (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",24.390244,24.390244,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.24±0.07; stderr=0.06789956540036612."],["epoch-frontier4-bg3JRdd33iWfvRoGSvyXXE","grok-4-6","Grok 4.6","Grok 4.6 (xhigh)","grok-4-6-xhigh","Grok 4.6 (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",31.707317,31.707317,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","verification_code:0.32±0.07; stderr=0.07357611282438223."],["epoch-frontier4-jvJPoRYkPH9Sb5MX252oHp","inkling","Inkling","Inkling (xhigh)","inkling-xhigh","Inkling (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",4.878049,4.878049,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","verification_code:0.05±0.03; stderr=0.03405912205797303."],["epoch-frontier4-DVD8G8DHNyzkowBzK7JvMt","inkling-small","Inkling-Small","Inkling Small (xhigh)","inkling-small-epoch-inkling-small-xhigh","Inkling Small (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",17.073171,17.073171,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-18","source-checked","verification_code:0.17±0.06; stderr=0.05949419959829495."],["epoch-frontier4-eVKA2eavdRAbSN6AnzyBxA","kimi-k2-6","Kimi K2.6","Kimi K2.6","kimi-k2-6-epoch-kimi-k2-6","Kimi K2.6","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",25.641026,25.641026,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.26±0.07; stderr=0.07083413480167725."],["epoch-frontier4-UTVSgirxQpTgmTHbgE9EUk","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code","kimi-k2-7-code-epoch-kimi-k2-7-code","Kimi K2.7 Code","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",12.195122,12.195122,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-17","source-checked","verification_code:0.12±0.05; stderr=0.051739520574625435."],["epoch-frontier4-gZERSAQwxBAevQzBsWfUie","kimi-k3","Kimi K3","Kimi K3 (Max)","kimi-k3-max","Kimi K3 (Max)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",39.02439,39.02439,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-17","2026-07-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-17","2026-09-01","2026-08-17","source-checked","verification_code:0.39±0.08; stderr=0.07712872341874095."],["epoch-frontier4-iR3Ut7aYEtz8dTYQFzq7rG","o3-mini","o3-mini","o3-mini (high)","o3-mini-high","o3-mini (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",0,0,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-18","source-checked","verification_code:0±0; stderr=0.0."],["epoch-frontier4-TomkQucNryhFGkaUnUGeLR","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",4.878049,4.878049,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.05±0.03; stderr=0.03405912205797303."],["epoch-frontier4-FHZCKbtfvpKGXU99vZFnXN","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max","qwen-3-7-max-epoch-qwen3-7-max","Qwen3.7 Max","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",34.146341,34.146341,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-17","source-checked","verification_code:0.34±0.07; stderr=0.07497768853141171."],["epoch-frontier4-PBRdMZyz4L4CqPpXyLD2Lk","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh)","benchlm-frontiermathv2tier4","FrontierMath v2 Tier 4","mathematics","Epoch AI","v2 / task implementation 2.0.0",46.341463,46.341463,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-04","2026-08-04","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-04","2026-09-01","2026-08-17","source-checked","verification_code:0.46±0.08; stderr=0.07884502378168105."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-07-21","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.069,2.3247,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-07-27","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.069,2.3247,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-08-01","claude-3-5-sonnet","Claude 3.5 Sonnet","Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.069,2.3247,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-07-21","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.903,6.6326,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-07-27","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.903,6.6326,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.903,6.6326,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",20.69,23.2472,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",20.69,23.2472,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",20.69,23.2472,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",43.793,49.2056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",43.793,49.2056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",43.793,49.2056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.241,53.0798,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.241,53.0798,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.241,53.0798,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",13.495,15.1629,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",13.495,15.1629,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",13.495,15.1629,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",32.4,36.4045,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",32.4,36.4045,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",32.4,36.4045,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.724,1.9371,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.724,1.9371,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.724,1.9371,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",22.1,24.8315,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",22.1,24.8315,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",22.1,24.8315,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.844,5.4427,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.844,5.4427,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.844,5.4427,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",14.138,15.8854,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",14.138,15.8854,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",14.138,15.8854,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",35.64,40.0449,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",35.64,40.0449,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",35.64,40.0449,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",37.6,42.2472,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",37.6,42.2472,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",37.6,42.2472,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",36.9,41.4607,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",36.9,41.4607,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",36.9,41.4607,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.819,4.291,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.819,4.291,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.819,4.291,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.439,2.7404,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.439,2.7404,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",2.439,2.7404,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tiers13-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",16.434,18.4652,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tiers13-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",16.434,18.4652,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-frontiermathv2tiers13-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",16.434,18.4652,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",33.448,37.582,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",33.448,37.582,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",33.448,37.582,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.517,6.1989,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.517,6.1989,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",5.517,6.1989,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.483,5.0371,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.483,5.0371,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",4.483,5.0371,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.034,1.1618,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.034,1.1618,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",1.034,1.1618,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.345,0.3876,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.345,0.3876,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.345,0.3876,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",31.034,34.8697,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",31.034,34.8697,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",31.034,34.8697,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",40.7,45.7303,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tiers13-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",50,56.1798,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tiers13-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",50,56.1798,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-frontiermathv2tiers13-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",50,56.1798,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.6,53.4831,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.6,53.4831,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",47.6,53.4831,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",28.28,31.7753,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",28.28,31.7753,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",28.28,31.7753,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",25.86,29.0562,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",25.86,29.0562,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",25.86,29.0562,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tiers13-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51,57.3034,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tiers13-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51,57.3034,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-frontiermathv2tiers13-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51,57.3034,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51.7,58.0899,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51.7,58.0899,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",51.7,58.0899,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",78.6,88.3146,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",78.6,88.3146,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",78.6,88.3146,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",89,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",89,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",89,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",84.9,95.3933,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",84.9,95.3933,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",84.9,95.3933,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-07-21","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.793,4.2618,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-07-27","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.793,4.2618,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-08-01","grok-3-beta","Grok 3 [Beta]","Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",3.793,4.2618,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tiers13-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",19.655,22.0843,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tiers13-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",19.655,22.0843,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-frontiermathv2tiers13-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",19.655,22.0843,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.404,24.0494,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.404,24.0494,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.404,24.0494,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",27.9,31.3483,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",27.9,31.3483,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",27.9,31.3483,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",38.966,43.782,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.69,0.7753,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.69,0.7753,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0.69,0.7753,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tiers13-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",39,43.8202,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tiers13-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",39,43.8202,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-frontiermathv2tiers13-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",39,43.8202,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-frontiermathv2tiers13-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",9.31,10.4607,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-frontiermathv2tiers13-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",9.31,10.4607,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-frontiermathv2tiers13-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",9.31,10.4607,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tiers13-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",18.685,20.9944,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tiers13-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",18.685,20.9944,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-frontiermathv2tiers13-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",18.685,20.9944,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-21--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-27--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-08-01--configuration--o4-mini-high","o4-mini","o4-mini","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","o4-mini-high","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-21","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-27","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-08-01","o4-mini-high","o4-mini (high)","Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",24.828,27.8966,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",23.103,25.9584,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",23.103,25.9584,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",23.103,25.9584,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-07-21","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",8.481,9.5292,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-07-27","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",8.481,9.5292,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-08-01","qwen3-235b-2507-reasoning","Qwen3 235B 2507 (Reasoning)","Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",8.481,9.5292,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-07-21","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",6.207,6.9742,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-07-27","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",6.207,6.9742,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-08-01","qwen3-5-flash","Qwen3.5 Flash","Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",6.207,6.9742,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-07-21","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.034,23.6337,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-07-27","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.034,23.6337,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-08-01","qwen3-5-plus","Qwen3.5 Plus","Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",21.034,23.6337,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",26.207,29.4461,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",26.207,29.4461,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","2026",26.207,29.4461,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["epoch-frontier13-E28uZiJvrPgKtawbJpoZQ8","claude-fable-5","Claude Fable 5","Claude Fable 5 (max)","claude-fable-5-max","Claude Fable 5 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",87.017544,87.017544,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-09","2026-06-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-09","2026-09-01","2026-08-17","source-checked","verification_code:0.87±0.02; stderr=0.019944477920133024."],["epoch-frontier13-5eRkmXCURH2pRbhJoukfnC","claude-opus-4-1","Claude Opus 4.1","Claude Opus 4.1 (32k thinking)","claude-opus-4-1-epoch-claude-opus-4-1-20250805-32k","Claude Opus 4.1 (32k thinking)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",12.631579,12.631579,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-18","source-checked","verification_code:0.13±0.02; stderr=0.0197127354633578."],["epoch-frontier13-mS5EwCRZVpvUvJEehPUDZu","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (32k thinking)","claude-opus-4-5-epoch-claude-opus-4-5-20251101-32k","Claude Opus 4.5 (32k thinking)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",34.385965,34.385965,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.34±0.03; stderr=0.028185763989072587."],["epoch-frontier13-kgJ2d6XbKj3dZutDcBYSKU","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (max)","claude-opus-4-6-max","Claude Opus 4.6 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",65.964912,65.964912,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.66±0.03; stderr=0.028116467881852295."],["epoch-frontier13-ZkNQ7GER7Mc9FSPmpUzgKH","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (max)","claude-opus-4-7-max","Claude Opus 4.7 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",70.175439,70.175439,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.7±0.03; stderr=0.027146911721220194."],["epoch-frontier13-GVmgtgKMwEqc54f3jA7auP","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max)","claude-opus-4-8-max","Claude Opus 4.8 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",80,80,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.8±0.02; stderr=0.023735633163877067."],["epoch-frontier13-ff7GeS7rdperjxSBZ7MHZY","claude-opus-5","Claude Opus 5","Claude Opus 5 (max)","claude-opus-5-max","Claude Opus 5 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",85.614035,85.614035,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-24","2026-07-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-24","2026-09-01","2026-08-17","source-checked","verification_code:0.86±0.02; stderr=0.020824894575343943."],["epoch-frontier13-7oUkeLfwMK3MtkSNHX6QPu","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (32k thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929-32k","Claude Sonnet 4.5 (32k thinking)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",23.859649,23.859649,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.24±0.03; stderr=0.02529183228411416."],["epoch-frontier13-V6YuGpcu8MCbJNXKoJMA6b","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max)","claude-sonnet-5-max","Claude Sonnet 5 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",65.614035,65.614035,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-30","2026-06-30","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-30","2026-09-01","2026-08-17","source-checked","verification_code:0.66±0.03; stderr=0.02818576398907258."],["epoch-frontier13-mKhVMEicZeGjdvJdTRd7mK","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731 (max)","deepseek-v4-flash-0731-epoch-deepseek-v4-flash-0731-max","DeepSeek V4 Flash 0731 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",57.54386,57.54386,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0.58±0.03; stderr=0.029329899789943818."],["epoch-frontier13-KUar3FaFjzq5SMKrEhtn8e","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek v4 (max)","deepseek-v4-pro-max","DeepSeek v4 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",45.263158,45.263158,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-17","2026-06-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-17","2026-09-01","2026-08-18","source-checked","verification_code:0.45±0.03; stderr=0.029536098269922092."],["epoch-frontier13-FkywQ2KS4AkLPzjSPmhHcS","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro","gemini-2-5-pro-epoch-gemini-2-5-pro","Gemini 2.5 Pro","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",24.561404,24.561404,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.25±0.03; stderr=0.0255425481026507."],["epoch-frontier13-EtgrDq6ei6kZe8K7QbXUe3","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview","gemini-3-flash-epoch-gemini-3-flash-preview","Gemini 3 Flash Preview","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",51.22807,51.22807,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.51±0.03; stderr=0.02966059084324674."],["epoch-frontier13-hVifTUdQZrFEYabgCRu8Qq","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview","gemini-3-1-pro-preview-epoch-gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",59.649123,59.649123,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.6±0.03; stderr=0.02911181956525014."],["epoch-frontier13-72DQVwbYiQW5pjhshjBaGU","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",62.807018,62.807018,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.63±0.03; stderr=0.028679753751074618."],["epoch-frontier13-8o2J9uQLKDCTiCFwYdkJRK","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Gemini 3.5 Flash-Lite (high)","gemini-3-5-flash-lite-livebench-2026-06-25-high","Gemini 3.5 Flash-Lite (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",25.964912,25.964912,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0.26±0.03; stderr=0.02601675082238553."],["epoch-frontier13-nb9LnDeeQyvRCqsbsTANGD","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high)","gemini-3-6-flash-high","Gemini 3.6 Flash (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",58.947368,58.947368,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","verification_code:0.59±0.03; stderr=0.02919063494391404."],["epoch-frontier13-axMdgNDC8UYYLnZYLHZpF5","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high)","gemini-3-7-flash-high","Gemini 3.7 Flash (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",71.578947,71.578947,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","verification_code:0.72±0.03; stderr=0.026764156649364653."],["epoch-frontier13-TTuCPacMDpNaCSs3Hpbdjp","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",59.205776,59.205776,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-19","2026-06-19","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-19","2026-09-01","2026-08-17","source-checked","verification_code:0.59±0.03; stderr=0.029581952519606228."],["epoch-frontier13-iXvqUzwxAiA3Rg3dwnutFs","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",68.77193,68.77193,"percent","higher","2.2.0","reference-only","direct","2026-08-25","2026-08-25","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-25","2026-09-01","2026-09-01","independently-verified","verification_code:0.69±0.03; stderr=0.027499133473298815; Epoch run iXvqUzwxAiA3Rg3dwnutFs. frontiermath:v2:tiers-1-3 is an existing reference-only protocol and cannot alter generic scoring."],["epoch-frontier13-QRDHTGX3rYrVCjVPRarjjt","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",55.438596,55.438596,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.55±0.03; stderr=0.029493504108183504."],["epoch-frontier13-YSR9WCR2SnJYvP7DJwX7YN","gpt-5-mini","GPT-5 mini","GPT-5 mini (high)","gpt-5-mini-high","GPT-5 mini (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",46.666667,46.666667,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.47±0.03; stderr=0.029603535719125714."],["epoch-frontier13-JH3PjDrbrQndEcc8fh5Qre","gpt-5-nano","GPT-5 nano","GPT-5 nano (high)","gpt-5-nano-high","GPT-5 nano (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",20,20,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-18","source-checked","verification_code:0.2±0.02; stderr=0.023735633163877067."],["epoch-frontier13-PeZ84GTQdSYQzDxRMWA2Ha","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",67.4,67.4,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.67±0.03; stderr=0.028."],["epoch-frontier13-dMJenf5xYWu7Pwn2U6w4dW","gpt-5-2","GPT-5.2","GPT-5.2 Pro (xhigh)","gpt-5-2-pro-epoch-gpt-5-2-pro-2025-12-11-xhigh","GPT-5.2 Pro (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",74,74,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-18","source-checked","verification_code:0.74±0.03; stderr=0.026349537704451795."],["epoch-frontier13-PaYEQ37A2n6giYBwZGBnXY","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",78.596491,78.596491,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.79±0.02; stderr=0.024338000553262393."],["epoch-frontier13-ZFVa8dw7oiir6PhcF8XZMo","gpt-5-4","GPT-5.4","GPT-5.4 Pro (xhigh)","gpt-5-4-pro-xhigh","GPT-5.4 Pro (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",82.45614,82.45614,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-18","source-checked","verification_code:0.82±0.02; stderr=0.022569134425265196."],["epoch-frontier13-jjFRrQvsud3XsgYufr9f8f","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",51.22807,51.22807,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.51±0.03; stderr=0.02966059084324674."],["epoch-frontier13-iKtopjTjyCWZqJaEjcc7ZL","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (high)","gpt-5-4-nano-epoch-gpt-5-4-nano-2026-03-17-high","GPT-5.4 nano (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",44.912281,44.912281,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.45±0.03; stderr=0.029515543245521678."],["epoch-frontier13-5FGXKvXQwGPtcCX2PNrLR5","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",85.263158,85.263158,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.85±0.02; stderr=0.021034091168852485."],["epoch-frontier13-LSKDetgzyVojru45w7RJhf","gpt-5-5","GPT-5.5","GPT-5.5 Pro (xhigh)","gpt-5-5-pro","GPT-5.5 Pro (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",87.719298,87.719298,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-12","2026-06-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-12","2026-09-01","2026-08-17","source-checked","verification_code:0.88±0.02; stderr=0.019476010341530077."],["epoch-frontier13-YLAk5J3Wd8EruVXee8bAQb","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",82.105263,82.105263,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.82±0.02; stderr=0.02274515950332741."],["epoch-frontier13-VBpT2WWYKBZtUxG9Z3svGh","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",89.122807,89.122807,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.89±0.02; stderr=0.018475392571140503."],["epoch-frontier13-J6evjTMgHkrun8txnEvzdN","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",85.964912,85.964912,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.86±0.02; stderr=0.020611471958164294."],["epoch-frontier13-FLDt53waLTeshC7PK9uVy2","grok-4-20","Grok 4.20","grok-4.20-0309-reasoning","grok-4-20-epoch-grok-4-20-0309-reasoning","grok-4.20-0309-reasoning","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",44.912281,44.912281,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","verification_code:0.45±0.03; stderr=0.029515543245521678."],["epoch-frontier13-bdsZjrG4ayzP8NAJg5q586","grok-4-3","Grok 4.3","grok-4.3_high","grok-4-3-epoch-grok-4-3-high","grok-4.3_high","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",42.807018,42.807018,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-17","2026-06-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-17","2026-09-01","2026-08-17","source-checked","verification_code:0.43±0.03; stderr=0.029360921879029937."],["epoch-frontier13-djbbEqvBW9qux3pZwcZN2i","grok-4-5","Grok 4.5","Grok 4.5 (high)","grok-4-5-aa-2-high","Grok 4.5 (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",57.192982,57.192982,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","verification_code:0.57±0.03; stderr=0.029360921879029937."],["epoch-frontier13-Zb2AKwa942D7dBTTcebAba","grok-4-6","Grok 4.6","Grok 4.6 (xhigh)","grok-4-6-xhigh","Grok 4.6 (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",65.964912,65.964912,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","verification_code:0.66±0.03; stderr=0.028116467881852295."],["epoch-frontier13-6KvQvuMMX5zAuteUc6pFcW","inkling","Inkling","Inkling (xhigh)","inkling-xhigh","Inkling (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",33.333333,33.333333,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","verification_code:0.33±0.03; stderr=0.02797271194322297."],["epoch-frontier13-B6bZbqickq8o2fQcgDmzMo","inkling-small","Inkling-Small","Inkling Small (xhigh)","inkling-small-epoch-inkling-small-xhigh","Inkling Small (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",46.315789,46.315789,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-18","source-checked","verification_code:0.46±0.03; stderr=0.02958888847874604."],["epoch-frontier13-BxrNfyMffkVUq4n7U5eK3Q","kimi-k2-6","Kimi K2.6","Kimi K2.6","kimi-k2-6-epoch-kimi-k2-6","Kimi K2.6","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",57.192982,57.192982,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-10","2026-06-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-10","2026-09-01","2026-08-17","source-checked","verification_code:0.57±0.03; stderr=0.029360921879029937."],["epoch-frontier13-X3RxRkt2pBUXYJjAQHsUo6","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code","kimi-k2-7-code-epoch-kimi-k2-7-code","Kimi K2.7 Code","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",54.035088,54.035088,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-17","source-checked","verification_code:0.54±0.03; stderr=0.02957276813514758."],["epoch-frontier13-bcntamvGsSH6QgETdoaSph","kimi-k3","Kimi K3","Kimi K3 (Max)","kimi-k3-max","Kimi K3 (Max)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",72.183099,72.183099,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-17","2026-07-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-17","2026-09-01","2026-08-17","source-checked","verification_code:0.72±0.03; stderr=0.026636607935112643."],["epoch-frontier13-mRJaJyBY2M2GGdbdMmT6iq","o3-mini","o3-mini","o3-mini (high)","o3-mini-high","o3-mini (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",18.596491,18.596491,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-18","source-checked","verification_code:0.19±0.02; stderr=0.023087552563757593."],["epoch-frontier13-dcmRi5JTa27EuwiQTdqr2B","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",36.140351,36.140351,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-11","2026-06-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-11","2026-09-01","2026-08-17","source-checked","verification_code:0.36±0.03; stderr=0.028506918644975."],["epoch-frontier13-RG2biwnmGybSVTVxxMdRWa","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max","qwen-3-7-max-epoch-qwen3-7-max","Qwen3.7 Max","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",64.561404,64.561404,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-13","2026-06-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-13","2026-09-01","2026-08-17","source-checked","verification_code:0.65±0.03; stderr=0.02838347520543562."],["epoch-frontier13-ABT89ieSWGnGj4RYMduHiG","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh)","benchlm-frontiermathv2tiers13","FrontierMath v2 Tiers 1-3","mathematics","Epoch AI","v2 / task implementation 2.0.0",74.736842,74.736842,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-04","2026-08-04","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-04","2026-09-01","2026-08-17","source-checked","verification_code:0.75±0.03; stderr=0.0257841025556124."],["benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",90.8,72.3077,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",90.8,72.3077,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",90.8,72.3077,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",92.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",92.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",92.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",86.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",86.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-gsm8k","Grade School Math 8K","mathematics","DeepSeek-AI","2026",86.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",92.9,32.3529,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",92.9,32.3529,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",92.9,32.3529,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2025-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",97.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2025-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",97.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2025-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",97.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",95.4,69.1176,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",95.4,69.1176,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",95.4,69.1176,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",94.8,60.2941,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",94.8,60.2941,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",94.8,60.2941,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",93.8,45.5882,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",93.8,45.5882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",93.8,45.5882,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",96.7,88.2353,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",96.7,88.2353,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",96.7,88.2353,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",90.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",90.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmt2025","Harvard-MIT Mathematics Tournament February 2025","mathematics","Harvard and MIT Mathematics Departments","2025",90.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",85.3,83.4595,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",85.3,83.4595,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",85.3,83.4595,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hmmtfeb2026-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",91.9,92.711,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",40.8,21.0821,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",40.8,21.0821,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",40.8,21.0821,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94.8,96.776,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hmmtfeb2026-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",94,95.6546,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",31.7,8.3263,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",31.7,8.3263,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",31.7,8.3263,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",95.2,97.3367,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2026-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",86.4,85.0014,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2026-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",86.4,85.0014,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtfeb2026-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",86.4,85.0014,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtfeb2026-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",82.6,79.6748,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtfeb2026-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",82.6,79.6748,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtfeb2026-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",82.6,79.6748,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtfeb2026-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.5,93.552,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtfeb2026-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.5,93.552,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtfeb2026-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.5,93.552,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-hmmtfeb2026-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",90.2,90.328,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.1,85.9826,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.1,85.9826,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.1,85.9826,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hmmtfeb2026-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.7,93.8324,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hmmtfeb2026-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.7,93.8324,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hmmtfeb2026-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.7,93.8324,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.9,82.8988,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.9,82.8988,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.9,82.8988,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",25.76,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",25.76,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",25.76,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.9,87.104,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.9,87.104,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.9,87.104,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.3,82.0578,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.3,82.0578,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",84.3,82.0578,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.8,86.9638,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.8,86.9638,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",87.8,86.9638,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",83.6,81.0765,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",83.6,81.0765,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",83.6,81.0765,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",97.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",97.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",97.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.9,94.1127,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.9,94.1127,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",92.9,94.1127,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-hmmtfeb2026-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",71.6,64.2557,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-hmmtfeb2026-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",71.6,64.2557,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-hmmtfeb2026-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtfeb2026","Harvard-MIT Mathematics Tournament February 2026","mathematics","Qwen","2026",71.6,64.2557,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",93.3,53.8462,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",93.3,53.8462,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",93.3,53.8462,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtnov2025-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",96.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtnov2025-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",96.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hmmtnov2025-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",96.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtnov2025-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94,62.8205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtnov2025-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94,62.8205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hmmtnov2025-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94,62.8205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtnov2025-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.4,67.9487,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtnov2025-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.4,67.9487,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hmmtnov2025-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.4,67.9487,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtnov2025-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",91.1,25.641,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtnov2025-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",91.1,25.641,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hmmtnov2025-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",91.1,25.641,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",92.7,46.1538,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",92.7,46.1538,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",92.7,46.1538,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",90.7,20.5128,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",90.7,20.5128,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",90.7,20.5128,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.6,70.5128,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.6,70.5128,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",94.6,70.5128,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",89.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",89.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-hmmtnov2025","Harvard-MIT Mathematics Tournament November 2025","mathematics","Qwen","2025",89.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-command-a-plus-hmmt-feb-2026-2602-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",73.5,73.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-hmmt-feb-2026-2602-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",94.7,94.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-hmmt-feb-2026-2602-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",61.4,61.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-hmmt-feb-2026-2602-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",62.9,62.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-hmmt-feb-2026-2602-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",68.9,68.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-hmmt-feb-2026-2602","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT","2602",93.9,93.9,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-848","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,85.3,85.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-850","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,40.8,40.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-851","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,31.7,31.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-847","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,86.4,86.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-849","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,82.6,82.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-844","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,92.5,92.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-846","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,87.1,87.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-845","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,87.8,87.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-842","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,97.1,97.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-843","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","hmmt-feb-2026","HMMT Feb 2026","mathematics","HMMT",null,92.9,92.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-imoanswerbench-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",85.1,91.042,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",41.9,12.0658,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",41.9,12.0658,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",41.9,12.0658,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88.4,97.075,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-imoanswerbench-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",88,96.3437,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",35.3,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",35.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",35.3,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",89.8,99.6344,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-longcat-2-0-benchlm-imoanswerbench-2026","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",81.8,81.8,"percent","higher","2.1.0","reference-only","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score for IMO-AnswerBench."],["benchlm-ref-qwen3-7-max-imoanswerbench-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",90,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-imoanswerbench-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",90,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-imoanswerbench-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",90,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-imoanswerbench-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",86,92.6874,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-imoanswerbench-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",86,92.6874,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-imoanswerbench-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",86,92.6874,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-imoanswerbench-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",59.3,43.8757,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-imoanswerbench-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",59.3,43.8757,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-imoanswerbench-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","benchlm-imoanswerbench","IMOAnswerBench","mathematics","DeepSeek-AI","2026",59.3,43.8757,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-imo2026-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-imo2026","International Mathematical Olympiad 2026","mathematics","Anthropic","2026",42,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-imo2026-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-imo2026","International Mathematical Olympiad 2026","mathematics","Anthropic","2026",42,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-ipho2025theory-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-ipho2025theory","International Physics Olympiad 2025 (Theory)","mathematics","Meta AI","2026",93.5,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-ipho2025theory-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-ipho2025theory","International Physics Olympiad 2025 (Theory)","mathematics","Meta AI","2026",93.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-ipho2025theory-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-ipho2025theory","International Physics Olympiad 2025 (Theory)","mathematics","Meta AI","2026",93.5,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",57.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",57.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",57.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",64.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",64.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-mathbenchmark","MATH","mathematics","DeepSeek-AI","2026",64.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-1929","gpt-5","GPT-5","Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.",null,"Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","math-500","MATH-500","mathematics","OpenAI / Hendrycks",null,99.4,99.4,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-math-500","aa-math-500","MATH-500 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/math-500","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis MATH-500 leaderboard page summary (top scores) checked 2026-07-16."],["evidence-2026-07-1929--configuration--gpt-5-high","gpt-5","GPT-5","Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","gpt-5-high","Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","math-500","MATH-500","mathematics","OpenAI / Hendrycks",null,99.4,99.4,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-math-500","aa-math-500","MATH-500 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/math-500","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis MATH-500 leaderboard page summary (top scores) checked 2026-07-16."],["evidence-2026-07-1931","grok-3-mini","Grok 3 mini","Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.",null,"Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","math-500","MATH-500","mathematics","OpenAI / Hendrycks",null,99.2,99.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-math-500","aa-math-500","MATH-500 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/math-500","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis MATH-500 leaderboard page summary (top scores) checked 2026-07-16."],["evidence-2026-07-1930","o3","o3","Artificial Analysis MATH-500 evaluation; reasoning setting as published as stated on the MATH-500 leaderboard summary.",null,"Artificial Analysis MATH-500 evaluation; reasoning setting as published as stated on the MATH-500 leaderboard summary.","math-500","MATH-500","mathematics","OpenAI / Hendrycks",null,99.2,99.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-math-500","aa-math-500","MATH-500 Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/math-500","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis MATH-500 leaderboard page summary (top scores) checked 2026-07-16."],["benchlm-ref-lfm2-5-8b-a1b-math500-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",88.76,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-math500-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",88.76,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-math500-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",88.76,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-math500-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",91.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-math500-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",91.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-math500-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-math500","MATH-500 Problem Set","mathematics","Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","2021",91.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmanswerbench-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",84,42.1488,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmanswerbench-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",84,42.1488,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmanswerbench-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",84,42.1488,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmanswerbench-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",82.5,29.7521,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmanswerbench-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",82.5,29.7521,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-mmanswerbench-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",82.5,29.7521,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mmanswerbench-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mmanswerbench-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-mmanswerbench-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mmanswerbench-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",91,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mmanswerbench-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",91,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-mmanswerbench-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",91,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmanswerbench-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",81.8,23.9669,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmanswerbench-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",81.8,23.9669,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmanswerbench-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",81.8,23.9669,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmanswerbench-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",86,58.6777,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmanswerbench-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",86,58.6777,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmanswerbench-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",86,58.6777,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmanswerbench-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.9,16.5289,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmanswerbench-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.9,16.5289,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmanswerbench-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.9,16.5289,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmanswerbench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.8,15.7025,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmanswerbench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.8,15.7025,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmanswerbench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",80.8,15.7025,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmanswerbench-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmanswerbench-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmanswerbench-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",83.8,40.4959,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",78.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",78.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmanswerbench","MMAnswerBench","mathematics","Qwen","2026",78.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-riemannbenchwithtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-riemannbenchwithtools","RiemannBench with tools","mathematics","Surge AI and Anthropic","2026",79,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-riemannbenchwithtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-riemannbenchwithtools","RiemannBench with tools","mathematics","Surge AI and Anthropic","2026",79,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-riemannbench-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-riemannbench","RiemannBench without tools","mathematics","Surge AI and Anthropic","2026",60,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-riemannbench-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-riemannbench","RiemannBench without tools","mathematics","Surge AI and Anthropic","2026",60,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-usamo2026-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",97.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-usamo2026-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",97.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-usamo2026-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",97.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-usamo2026-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",96.7,92.4306,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-usamo2026-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",96.7,92.4306,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-usamo2026-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",96.7,92.4306,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-usamo2026-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",85.71,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-usamo2026-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",85.71,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-usamo2026-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals","2026",85.71,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-772","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals",null,99.8,99.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-773","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals",null,96.7,96.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-774","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","usamo-2026","USAMO 2026","mathematics","MAA / public contest evals",null,85.71,85.71,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",88.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",88.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",88.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",92.7,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",92.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ai2dtest","AI2D test split","multimodal","Qwen","2026",92.7,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-babyvision-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvision","BabyVision","multimodal","Meta AI","2026",76.3,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-babyvision-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvision","BabyVision","multimodal","Meta AI","2026",76.3,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-babyvision-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvision","BabyVision","multimodal","Meta AI","2026",76.3,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-babyvision-with-ci-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-babyvision","BabyVision","multimodal","Meta AI","With CI",70.4,70.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-babyvision-with-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-babyvision","BabyVision","multimodal","Meta AI","With CI",85.6,85.6,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-08-15-claude-opus-4-6-benchlm-babyvision-without-ci-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-babyvision","BabyVision","multimodal","Meta AI","Without CI",12.6,12.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-babyvision-without-ci-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-babyvision","BabyVision","multimodal","Meta AI","Without CI",28.9,28.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-babyvision-without-ci-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-babyvision","BabyVision","multimodal","Meta AI","Without CI",64.7,64.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-babyvision-without-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-babyvision","BabyVision","multimodal","Meta AI","Without CI",65.7,65.7,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-kimi-3-babyvisionpython-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvisionpython","BabyVision with Python","multimodal","Moonshot AI","2026",85.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-babyvisionpython-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvisionpython","BabyVision with Python","multimodal","Moonshot AI","2026",85.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-babyvisionpython-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-babyvisionpython","BabyVision with Python","multimodal","Moonshot AI","2026",85.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-benchcadvision2codewithtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-benchcadvision2codewithtools","BenchCAD Vision2Code voxel IoU with tools","multimodal","Zhang et al. and Anthropic","2026",0.821,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-benchcadvision2codewithtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-benchcadvision2codewithtools","BenchCAD Vision2Code voxel IoU with tools","multimodal","Zhang et al. and Anthropic","2026",0.821,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-benchcadvision2code-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-benchcadvision2code","BenchCAD Vision2Code voxel IoU without tools","multimodal","Zhang et al. and Anthropic","2026",0.366,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-benchcadvision2code-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-benchcadvision2code","BenchCAD Vision2Code voxel IoU without tools","multimodal","Zhang et al. and Anthropic","2026",0.366,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-396","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",79.7,79.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-391","claude-mythos-5","Claude Mythos 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",92.8,92.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-524","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",59.3,59.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-457","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",75.4,75.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-406","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",66.2,66.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-512","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",84.4,84.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-497","claude-sonnet-5","Claude Sonnet 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",75.1,75.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-551","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",71.4,71.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-541","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",95,95,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-470","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",80.3,80.3,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-424","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",75.6,75.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-546","gemma-4-31b","Gemma 4 31B","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",67.6,67.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-492","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",56.6,56.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-518","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",91.8,91.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-487","gpt-5-3-codex","GPT-5.3-Codex","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",91.4,91.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-430","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",60,60,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-481","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",53.9,53.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-418","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",58.3,58.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-475","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",69.5,69.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-401","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",75.4,75.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-435","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",72.5,72.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-559","grok-4-1","Grok 4.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",93.1,93.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-501","grok-4-3","Grok 4.3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",69.2,69.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-538","mimo-v2-omni","MiMo-V2-Omni","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",67.5,67.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-451","minimax-m3","MiniMax M3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",49.6,49.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-444","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",74.6,74.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-411","muse-spark-1-1","Muse Spark 1.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",75.3,75.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-506","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-multimodal","BenchLM Multimodal prior","multimodal","BenchLM","bench-align-v5.1",68.4,68.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (multimodalGrounded) used to estimate missing category coverage under methodology 1.3.0."],["benchlm-ref-claude-fable-blueprintbench2-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",38.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-blueprintbench2-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",38.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-blueprintbench2-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",38.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",33.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",33.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","2026",33.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-359","claude-fable-5","Claude Fable 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","v2",38.6,38.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-019","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","v2",26.5,26.5,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-010","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","v2",33.6,33.6,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-360","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","v2",33.6,33.6,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-029","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","blueprint-bench-2","Blueprint-Bench 2","multimodal","Google","v2",36.2,36.2,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["benchlm-ref-qwen3-6-27b-ccocr-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-ccocr-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-ccocr-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-ccocr","CC-OCR","multimodal","Qwen","2026",81.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-chartographywithtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-chartographywithtools","Chartography with image and code tools","multimodal","Surge AI and Anthropic","2026",83,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-chartographywithtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-chartographywithtools","Chartography with image and code tools","multimodal","Surge AI and Anthropic","2026",83,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-chartography-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-chartography","Chartography without tools","multimodal","Surge AI and Anthropic","2026",29.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-chartography-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-chartography","Chartography without tools","multimodal","Surge AI and Anthropic","2026",29.6,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-flash-vision-exp-chartography-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","Provider chart does not specify tool use or evaluation settings",null,"Provider chart does not specify tool use or evaluation settings","benchlm-chartography","Chartography without tools","multimodal","Surge AI and Anthropic","2026",64.3,64.3,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value; exact tool policy, evaluation system, harness and model configuration are unavailable in the provider chart."],["benchlm-ref-claude-mythos-5-charxiv-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",93.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-charxiv-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",93.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-charxiv-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",93.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-charxiv-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",68.5,38.7255,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-charxiv-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",68.5,38.7255,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-charxiv-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",68.5,38.7255,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxiv-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",91,93.8725,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxiv-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",91,93.8725,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxiv-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",91,93.8725,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxiv-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",89.9,91.1765,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxiv-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",89.9,91.1765,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxiv-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",89.9,91.1765,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-charxiv-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.4,60.5392,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-charxiv-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.4,60.5392,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-charxiv-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.4,60.5392,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxiv-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.3,87.2549,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxiv-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.3,87.2549,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxiv-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.3,87.2549,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-charxiv-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",52.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-charxiv-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",52.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-charxiv-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",52.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-charxiv-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.4,70.3431,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-charxiv-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.4,70.3431,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-charxiv-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.4,70.3431,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",73.2,50.2451,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",73.2,50.2451,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",73.2,50.2451,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-charxiv-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.2,67.402,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-charxiv-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.2,67.402,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-charxiv-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.2,67.402,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-charxiv-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",84.2,77.2059,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-charxiv-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",84.2,77.2059,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-charxiv-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",84.2,77.2059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-charxiv-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.1,72.0588,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-charxiv-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.1,72.0588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-charxiv-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.1,72.0588,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-charxiv-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.8,73.7745,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-charxiv-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.8,73.7745,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-charxiv-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82.8,73.7745,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-charxiv-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",60.9,20.098,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-charxiv-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",60.9,20.098,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-charxiv-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",60.9,20.098,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxiv-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82,71.8137,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxiv-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82,71.8137,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxiv-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",82,71.8137,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-charxiv-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.3,70.098,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-charxiv-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.4,67.8922,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-charxiv-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.4,67.8922,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-charxiv-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.4,67.8922,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxiv-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",91.3,94.6078,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxiv-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",91.3,94.6078,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-charxiv-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81,69.3627,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-charxiv-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81,69.3627,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-charxiv-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81,69.3627,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-charxiv-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.4,82.598,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-charxiv-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.4,82.598,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-charxiv-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.4,82.598,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-charxiv-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.4,87.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-charxiv-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.4,87.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-charxiv-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",88.4,87.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",76.25,57.7206,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",76.25,57.7206,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",76.25,57.7206,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-charxiv-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.8,68.8725,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-charxiv-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.8,68.8725,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-charxiv-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",80.8,68.8725,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.2,60.049,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.2,60.049,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",77.2,60.049,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-charxiv-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78.4,62.9902,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-charxiv-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78.4,62.9902,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-charxiv-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78.4,62.9902,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-charxiv-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.5,70.5882,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-charxiv-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.5,70.5882,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-charxiv-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",81.5,70.5882,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78,62.0098,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78,62.0098,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",78,62.0098,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-charxiv-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.9,81.3725,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-charxiv-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.9,81.3725,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-charxiv-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.9,81.3725,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-charxiv-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.1,79.4118,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-charxiv-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.1,79.4118,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-charxiv-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",85.1,79.4118,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-charxiv-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.6,83.0882,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-charxiv-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.6,83.0882,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-charxiv-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","charxiv","CharXiv","multimodal","CharXiv","2024",86.6,83.0882,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxiv-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","charxiv","CharXiv","multimodal","CharXiv","CharXiv (RQ) with Python",91.3,94.6078,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports CharXiv (RQ) as 84.8 without tools and 91.3 with Python; this carrier is the with-Python track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["evidence-2026-07-232","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",93.5,93.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-646","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",68.5,68.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-233","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",89.9,89.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-charxiv:charxiv:2","claude-opus-5","Claude Opus 5","Claude Opus 5 (max) as published in Meta's comparison table","claude-opus-5-max","Claude Opus 5 (max) as published in Meta's comparison table","charxiv","CharXiv","multimodal","CharXiv","May 2026",89.3,89.3,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 1,000 validation questions; binary accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["evidence-2026-07-645","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",77.4,77.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-2033","claude-sonnet-5","Claude Sonnet 5","CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.",null,"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","charxiv","CharXiv","multimodal","CharXiv","May 2026",77,77,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Claude Sonnet 5 CharXiv no-tools comparison score."],["google-gemini-37-eval-charxiv-no-tools-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","charxiv","CharXiv","multimodal","CharXiv","May 2026",77,77,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-charxiv-with-tools-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.3,88.3,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-07-234","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.3,88.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-644","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",81.4,81.4,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-239","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",73.2,73.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-015","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","charxiv","CharXiv","multimodal","CharXiv","May 2026",83.3,83.3,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-238","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",80.2,80.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-006","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","charxiv","CharXiv","multimodal","CharXiv","May 2026",84.2,84.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-236","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",84.2,84.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["google-gemini-37-eval-charxiv-no-tools-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","charxiv","CharXiv","multimodal","CharXiv","May 2026",85.2,85.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-charxiv-with-tools-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","charxiv","CharXiv","multimodal","CharXiv","May 2026",89.4,89.4,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-07-1995","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning with search and code-execution tools.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning with search and code-execution tools.","charxiv","CharXiv","multimodal","CharXiv","May 2026",89.4,89.4,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","CharXiv Reasoning with tools; retained as reference configuration alongside no-tools row."],["evidence-2026-07-1994","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning without tools; self-computed.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning without tools; self-computed.","charxiv","CharXiv","multimodal","CharXiv","May 2026",85.2,85.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","CharXiv Reasoning (no tools) from Google evaluation table."],["google-gemini-37-eval-charxiv-no-tools-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","charxiv","CharXiv","multimodal","CharXiv","May 2026",84.5,84.5,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-charxiv-with-tools-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.7,88.7,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-charxiv:charxiv:4","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high) as published in Meta's comparison table","gemini-3-7-flash-high","Gemini 3.7 Flash (high) as published in Meta's comparison table","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.7,88.7,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 1,000 validation questions; binary accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["new-model:aa-glm-5-3-flash-2026-08-27:glm-5-3-flash:charxiv:cell:glm-5-3-flash-launch:charxiv:0","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max effort, documented default)","glm-5-3-flash-max","GLM-5.3-Flash (max effort, documented default)","charxiv","CharXiv","multimodal","CharXiv","May 2026",89.4,89.4,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-26",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-26","2026-08-27","2026-08-27","source-checked","Code tools enabled; 256K context."],["evidence-2026-07-237","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",82.8,82.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-025","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","charxiv","CharXiv","multimodal","CharXiv","May 2026",84.1,84.1,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-2020","gpt-5-6-luna","GPT-5.6 Luna","CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.",null,"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","charxiv","CharXiv","multimodal","CharXiv","May 2026",82.7,82.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited GPT-5.6 Luna CharXiv comparison score."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-charxiv:charxiv:1","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max) as published in Meta's comparison table","gpt-5-6-sol-max","GPT-5.6 Sol (max) as published in Meta's comparison table","charxiv","CharXiv","multimodal","CharXiv","May 2026",84,84,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 1,000 validation questions; binary accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["google-gemini-37-eval-charxiv-no-tools-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","charxiv","CharXiv","multimodal","CharXiv","May 2026",85.9,85.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-07-2026","grok-4-5","Grok 4.5","CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.",null,"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","charxiv","CharXiv","multimodal","CharXiv","May 2026",81.6,81.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","Google-cited Grok 4.5 CharXiv comparison score."],["evidence-2026-07-1944","kimi-k3","Kimi K3","CharXiv Reasoning with Python; max reasoning; average of three runs.",null,"CharXiv Reasoning with Python; max reasoning; average of three runs.","charxiv","CharXiv","multimodal","CharXiv","May 2026",91.3,91.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 91.3 on CharXiv Reasoning with Python; 84.8 without tools also published."],["evidence-2026-07-1944--configuration--kimi-k3-max","kimi-k3","Kimi K3","CharXiv Reasoning with Python; max reasoning; average of three runs.","kimi-k3-max","CharXiv Reasoning with Python; max reasoning; average of three runs.","charxiv","CharXiv","multimodal","CharXiv","May 2026",91.3,91.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 91.3 on CharXiv Reasoning with Python; 84.8 without tools also published."],["evidence-2026-07-642","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",86.4,86.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-641","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.4,88.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-charxiv:charxiv:3","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.4,88.4,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 1,000 validation questions; binary accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["evidence-2026-07-1912","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (CharXiv Reasoning).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (CharXiv Reasoning).","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.4,88.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for CharXiv Reasoning; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1912--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (CharXiv Reasoning).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (CharXiv Reasoning).","charxiv","CharXiv","multimodal","CharXiv","May 2026",88.4,88.4,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for CharXiv Reasoning; retained with provider-reported provenance via Meta evaluation report."],["new-model:aa-muse-spark-1-2-2026-08-27:muse-spark-1-2:charxiv:cell:muse-spark-1-2-charxiv:charxiv:0","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","charxiv","CharXiv","multimodal","CharXiv","May 2026",87.6,87.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-20",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","source-checked","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90.; 1,000 validation questions; binary accuracy is reported with GPT-OSS-120B high as judge."],["evidence-2026-07-643","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",81.5,81.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-235","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","charxiv","CharXiv","multimodal","CharXiv","May 2026",85.9,85.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-muse-glimmer-30b-charxiv-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","charxiv","CharXiv","multimodal","CharXiv","Reasoning / validation set",78.8,78.8,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported chart-reasoning success rate with gpt-oss-120b semantic and mathematical judging."],["evidence-2026-08-muse-glimmer-30b-charxiv-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","charxiv","CharXiv","multimodal","CharXiv","Reasoning / validation set",78.8,78.8,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Provider-reported chart-reasoning success rate with gpt-oss-120b semantic and mathematical judging."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:charxiv-with-ci:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","charxiv","CharXiv","multimodal","CharXiv","RQ With CI",85.9,85.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-charxiv-rq-with-ci-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","charxiv","CharXiv","multimodal","CharXiv","RQ With CI",85.9,85.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:charxiv-with-ci:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","charxiv","CharXiv","multimodal","CharXiv","RQ With CI",90.2,90.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-charxiv-rq-with-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","charxiv","CharXiv","multimodal","CharXiv","RQ With CI",90.2,90.2,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:charxiv:cell:vision:charxiv-with-ci:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","charxiv","CharXiv","multimodal","CharXiv","RQ With CI",90.6,90.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:charxiv-no-ci:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",66,66,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-charxiv-rq-without-ci-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",66,66,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-charxiv-rq-without-ci-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",78.4,78.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:charxiv-no-ci:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",85.8,85.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-charxiv-rq-without-ci-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",85.8,85.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:charxiv-no-ci:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",83.7,83.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-charxiv-rq-without-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","charxiv","CharXiv","multimodal","CharXiv","RQ Without CI",83.7,83.7,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-claude-mythos-5-charxivnotools-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",88.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-charxivnotools-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",88.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-charxivnotools-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",88.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxivnotools-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",82.1,42.8571,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxivnotools-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",82.1,42.8571,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-charxivnotools-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",82.1,42.8571,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxivnotools-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",80.5,29.4118,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxivnotools-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",80.5,29.4118,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-charxivnotools-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",80.5,29.4118,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxivnotools-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",77,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxivnotools-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",77,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-charxivnotools-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",77,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxivnotools-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",78.1,9.2437,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxivnotools-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",78.1,9.2437,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-charxivnotools-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",78.1,9.2437,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-charxivnotools-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",77.4,3.3613,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxivnotools-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",84.8,65.5462,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxivnotools-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",84.8,65.5462,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-charxivnotools-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-charxivnotools","CharXiv Reasoning without tools","multimodal","CharXiv authors","2024",84.8,65.5462,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-countbench-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",73.31,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-countbench-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",73.31,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-countbench-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",73.31,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-countbench-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",97.8,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-countbench-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",97.8,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-countbench-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-countbench","CountBench","multimodal","Qwen","2026",97.8,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-designarenawebsite-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1175,65.1815,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-designarenawebsite-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1175,65.1815,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-designarenawebsite-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1170,65.6146,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-designarenawebsite-2026-07-21","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1207,70.462,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-designarenawebsite-2026-07-27","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1207,70.462,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-designarenawebsite-2026-08-01","claude-4-1-opus","Claude 4.1 Opus","Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1202,70.9302,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-designarenawebsite-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1332,91.0891,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-designarenawebsite-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1332,91.0891,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-designarenawebsite-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1324,91.196,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-07-21","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-07-27","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-08-01","claude-haiku-4-5","Claude Haiku 4.5","Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1147,61.794,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-07-21","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-07-27","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-08-01","claude-haiku-4-5-thinking","Claude Haiku 4.5 Thinking","Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1147,61.794,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-designarenawebsite-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1277,82.0132,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-designarenawebsite-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1277,82.0132,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-designarenawebsite-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1272,82.5581,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1277,82.0132,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1277,82.0132,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1272,82.5581,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-designarenawebsite-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-designarenawebsite-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-designarenawebsite-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1319,90.3654,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-designarenawebsite-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-designarenawebsite-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-designarenawebsite-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1319,90.3654,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-designarenawebsite-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-designarenawebsite-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-designarenawebsite-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1320,90.5316,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-designarenawebsite-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-designarenawebsite-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-designarenawebsite-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1320,90.5316,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-designarenawebsite-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1270,80.8581,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-designarenawebsite-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1270,80.8581,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-designarenawebsite-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1268,81.8937,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-designarenawebsite-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1341,94.0199,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,72.4422,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,72.4422,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1215,73.0897,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-07-21","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,72.4422,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-07-27","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,72.4422,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-08-01","claude-sonnet-4-5-thinking","Claude Sonnet 4.5 Thinking","Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1215,73.0897,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1314,88.1188,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1314,88.1188,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1310,88.8704,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-designarenawebsite-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1314,88.1188,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-designarenawebsite-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1314,88.1188,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-designarenawebsite-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1314,89.5349,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-designarenawebsite-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1150,61.0561,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-designarenawebsite-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1150,61.0561,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-designarenawebsite-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1145,61.4618,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-designarenawebsite-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-designarenawebsite-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-designarenawebsite-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1147,61.794,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-designarenawebsite-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-designarenawebsite-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1152,61.3861,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-designarenawebsite-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1147,61.794,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-designarenawebsite-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1204,69.967,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-designarenawebsite-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1204,69.967,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-designarenawebsite-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1200,70.598,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-07-21","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1204,69.967,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-07-27","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1204,69.967,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-08-01","deepseek-v3-2-thinking","DeepSeek V3.2 (Thinking)","Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1200,70.598,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1233,76.0797,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-designarenawebsite-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1233,76.0797,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1233,76.0797,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1233,76.0797,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1238,75.5776,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1233,76.0797,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1260,80.5648,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-designarenawebsite-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1260,80.5648,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1260,80.5648,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1260,80.5648,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1264,79.868,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1260,80.5648,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1145,60.231,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1145,60.231,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1140,60.6312,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1197,68.8119,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1197,68.8119,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1192,69.2691,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-designarenawebsite-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1226,73.5974,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-designarenawebsite-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1226,73.5974,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-designarenawebsite-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1221,74.0864,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1281,82.6733,"elo","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1281,82.6733,"elo","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1276,83.2226,"elo","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1285,83.3333,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1285,83.3333,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1283,84.3854,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-designarenawebsite-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1319,90.3654,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-designarenawebsite-2026-07-21","glm-4-5","GLM-4.5","Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1200,69.3069,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-designarenawebsite-2026-07-27","glm-4-5","GLM-4.5","Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1200,69.3069,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-designarenawebsite-2026-08-01","glm-4-5","GLM-4.5","Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1195,69.7674,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-designarenawebsite-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1176,65.3465,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-designarenawebsite-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1176,65.3465,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-designarenawebsite-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1171,65.7807,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-designarenawebsite-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1255,78.3828,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-designarenawebsite-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1255,78.3828,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-designarenawebsite-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1251,79.0698,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-designarenawebsite-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1278,82.1782,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-designarenawebsite-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1278,82.1782,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-designarenawebsite-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1274,82.8904,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-designarenawebsite-2026-07-21","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1278,82.1782,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-designarenawebsite-2026-07-27","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1278,82.1782,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-reasoning-designarenawebsite-2026-08-01","glm-5-reasoning","GLM-5 (Reasoning)","Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1274,82.8904,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-designarenawebsite-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1301,85.9736,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-designarenawebsite-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1301,85.9736,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-designarenawebsite-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1296,86.5449,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-designarenawebsite-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1305,86.6337,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-designarenawebsite-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1305,86.6337,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-designarenawebsite-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1299,87.0432,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-designarenawebsite-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1340,92.4092,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-designarenawebsite-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1340,92.4092,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-designarenawebsite-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1341,94.0199,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-designarenawebsite-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1258,78.8779,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-designarenawebsite-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1258,78.8779,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-designarenawebsite-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1255,79.7342,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-designarenawebsite-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1068,47.5248,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-designarenawebsite-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1068,47.5248,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-designarenawebsite-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1063,47.8405,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1027,40.7591,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1027,40.7591,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1022,41.0299,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1003,36.7987,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1003,36.7987,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",998,37.0432,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-designarenawebsite-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",861,13.3663,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-designarenawebsite-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",861,13.3663,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-designarenawebsite-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",856,13.4551,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1209,72.093,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1209,72.093,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-designarenawebsite-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1209,72.093,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1214,71.6172,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-designarenawebsite-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1209,72.093,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-designarenawebsite-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1217,72.1122,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-designarenawebsite-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1217,72.1122,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-designarenawebsite-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1212,72.5914,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1191,67.8218,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1191,67.8218,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1186,68.2724,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-designarenawebsite-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1224,73.2673,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-designarenawebsite-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1224,73.2673,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-designarenawebsite-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,73.7542,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1193,68.1518,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1193,68.1518,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1187,68.4385,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-designarenawebsite-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1250,77.5578,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-designarenawebsite-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1250,77.5578,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-designarenawebsite-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1245,78.0731,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-designarenawebsite-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1282,82.8383,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-designarenawebsite-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1282,82.8383,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-designarenawebsite-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1280,83.887,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-designarenawebsite-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",998,35.9736,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-designarenawebsite-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",998,35.9736,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-designarenawebsite-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",993,36.2126,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-designarenawebsite-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",882,16.8317,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-designarenawebsite-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",882,16.8317,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-designarenawebsite-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",877,16.9435,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-designarenawebsite-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1257,78.7129,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-designarenawebsite-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1257,78.7129,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-designarenawebsite-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1252,79.2359,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-designarenawebsite-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1225,73.4323,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-designarenawebsite-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1225,73.4323,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-designarenawebsite-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1219,73.7542,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-designarenawebsite-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-designarenawebsite-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1325,89.934,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-designarenawebsite-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1321,90.6977,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-designarenawebsite-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1229,74.0924,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-designarenawebsite-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1229,74.0924,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-designarenawebsite-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1206,71.5947,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-designarenawebsite-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1080,49.505,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-designarenawebsite-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1080,49.505,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-designarenawebsite-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1075,49.8339,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-designarenawebsite-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1279,82.3432,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-designarenawebsite-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1279,82.3432,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-designarenawebsite-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1275,83.0565,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-designarenawebsite-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1279,82.3432,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-designarenawebsite-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1279,82.3432,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-designarenawebsite-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1275,83.0565,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-designarenawebsite-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1306,86.7987,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-designarenawebsite-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1306,86.7987,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-designarenawebsite-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1302,87.5415,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1302,86.1386,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1302,86.1386,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1300,87.2093,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-designarenawebsite-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1386,100,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-designarenawebsite-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1386,100,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-designarenawebsite-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1377,100,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-designarenawebsite-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",901,19.967,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-designarenawebsite-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",901,19.967,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-designarenawebsite-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",896,20.0997,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-designarenawebsite-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",780,0,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-designarenawebsite-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",780,0,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-designarenawebsite-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",775,0,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-designarenawebsite-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1291,84.3234,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-designarenawebsite-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1291,84.3234,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-designarenawebsite-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1295,86.3787,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1298,85.4785,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1298,85.4785,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1309,88.7043,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-designarenawebsite-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1275,81.6832,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-designarenawebsite-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1275,81.6832,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-designarenawebsite-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1271,82.392,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-designarenawebsite-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1289,83.9934,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-designarenawebsite-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1289,83.9934,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-designarenawebsite-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1286,84.8837,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-designarenawebsite-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1109,54.2904,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-designarenawebsite-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1109,54.2904,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-designarenawebsite-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1104,54.6512,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-designarenawebsite-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1299,85.6436,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-designarenawebsite-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1299,85.6436,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-designarenawebsite-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1291,85.7143,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1129,57.5908,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1129,57.5908,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1127,58.4718,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-designarenawebsite-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1067,47.3597,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-designarenawebsite-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1067,47.3597,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-designarenawebsite-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1061,47.5083,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-designarenawebsite-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1148,60.7261,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-designarenawebsite-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1148,60.7261,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-designarenawebsite-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1144,61.2957,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-designarenawebsite-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1249,77.3927,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-designarenawebsite-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1249,77.3927,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-designarenawebsite-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1265,81.3953,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-designarenawebsite-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1293,84.6535,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-designarenawebsite-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1293,84.6535,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-designarenawebsite-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1303,87.7076,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-designarenawebsite-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1288,83.8284,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-designarenawebsite-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1288,83.8284,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-designarenawebsite-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1297,86.711,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-designarenawebsite-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1211,71.1221,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-designarenawebsite-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1211,71.1221,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-designarenawebsite-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1212,72.5914,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-designarenawebsite-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1165,63.5314,"elo","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-designarenawebsite-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1165,63.5314,"elo","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-designarenawebsite-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1162,64.2857,"elo","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-designarenawebsite-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1165,63.5314,"elo","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-designarenawebsite-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1165,63.5314,"elo","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-designarenawebsite-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","benchlm-designarenawebsite","Design Arena Website Elo","multimodal","Design Arena","2026",1162,64.2857,"elo","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-dynamath-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-dynamath","DynaMath","multimodal","Qwen","2026",85.6,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-dynamath-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-dynamath","DynaMath","multimodal","Qwen","2026",85.6,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-dynamath-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-dynamath","DynaMath","multimodal","Qwen","2026",85.6,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:erqa:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-erqa","ERQA","multimodal","Qwen","2026",40.8,40.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-erqa-2026-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-erqa","ERQA","multimodal","Qwen","2026",40.8,40.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-claude-opus-4-6-erqa-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",51.6,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-erqa-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",51.6,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-erqa-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",51.6,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-erqa-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.4,97.8022,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-erqa-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.4,97.8022,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-erqa-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.4,97.8022,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-erqa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",65.4,75.8242,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-erqa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",65.4,75.8242,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-erqa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",65.4,75.8242,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-erqa-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",54.1,13.7363,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-erqa-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",54.1,13.7363,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-erqa-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",54.1,13.7363,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-erqa-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",64.7,71.978,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-erqa-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",64.7,71.978,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-erqa-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",64.7,71.978,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-erqa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",62.5,59.8901,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-erqa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",62.5,59.8901,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-erqa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",62.5,59.8901,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-erqa-2026-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-erqa","ERQA","multimodal","Qwen","2026",62.5,62.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-qwen3-7-plus-erqa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.8,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-erqa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.8,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-erqa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.8,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:erqa:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.8,69.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-erqa-2026-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-erqa","ERQA","multimodal","Qwen","2026",69.8,69.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:erqa:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-erqa","ERQA","multimodal","Qwen","2026",65.5,65.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-erqa-2026","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-erqa","ERQA","multimodal","Qwen","2026",65.5,65.5,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-erqa:cell:vision:erqa:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-erqa","ERQA","multimodal","Qwen","2026",72.3,72.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-claude-opus-5-gdppdfwithtools-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdppdfwithtools","GDP.pdf mean criteria pass rate with tools","multimodal","Surge AI and Anthropic","2026",85.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdppdfwithtools-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdppdfwithtools","GDP.pdf mean criteria pass rate with tools","multimodal","Surge AI and Anthropic","2026",85.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdppdf-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","2026",83.4,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-gdppdf-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","2026",83.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-gdp-pdf-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","August 2026, no tools",28,28,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdp-pdf-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","August 2026, no tools",22,22,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdp-pdf-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","August 2026, no tools",34,34,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-gdp-pdf-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","August 2026, no tools",24.7,24.7,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-gdp-pdf-muse-spark-1-2-2026-08-13","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2","muse-spark-1-2-xhigh","Muse Spark 1.2","benchlm-gdppdf","GDP.pdf mean criteria pass rate without tools","multimodal","Surge AI and Anthropic","August 2026, no tools",16,16,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-07-21","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",98.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-07-27","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",98.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-08-01","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractjsonvalidity","Liquid image-to-JSON extraction JSON validity","multimodal","Liquid AI","2026",98.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",99.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-07-21","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",98.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-07-27","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",98.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-08-01","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractschemaf1","Liquid image-to-JSON extraction schema consistency F1","multimodal","Liquid AI","2026",98.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",90.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",90.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",90.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-07-21","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",84.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-07-27","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",84.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-08-01","lfm2-5-vl-450m-extract","LFM2.5-VL-450M-Extract","Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","benchlm-liquidextractvlmjudge","Liquid image-to-JSON extraction VLM judge score","multimodal","Liquid AI","2026",84.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:lvbench:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","lvbench","LVBench","multimodal","LVBench","August 2026",63,63,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["google-gemini-37-eval-lvbench-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","lvbench","LVBench","multimodal","LVBench","August 2026",68.5,68.5,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-lvbench-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","lvbench","LVBench","multimodal","LVBench","August 2026",84.2,84.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-lvbench-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","lvbench","LVBench","multimodal","LVBench","August 2026",85.4,85.4,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-lvbench-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","lvbench","LVBench","multimodal","LVBench","August 2026",78.9,78.9,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:lvbench:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","lvbench","LVBench","multimodal","LVBench","August 2026",76.2,76.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:lvbench:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","lvbench","LVBench","multimodal","LVBench","August 2026",72.4,72.4,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:lvbench:cell:vision:lvbench:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","lvbench","LVBench","multimodal","LVBench","August 2026",76.6,76.6,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-command-a-plus-mmmu-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",75.1,79.5612,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-mmmu-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",75.1,79.5612,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-mmmu-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",75.1,79.5612,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmmu-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",32.67,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmmu-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",32.67,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-mmmu-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",32.67,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",70.8,71.4982,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",70.8,71.4982,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",70.8,71.4982,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmmu-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.3,93.0621,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmmu-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.3,93.0621,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmmu-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.3,93.0621,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",83.9,96.0623,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",83.9,96.0623,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",83.9,96.0623,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.4,91.3745,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.4,91.3745,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.4,91.3745,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmu-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.9,94.1871,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmu-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.9,94.1871,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmu-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",82.9,94.1871,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmu-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",86,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmu-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",86,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmu-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",86,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.7,91.937,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.7,91.937,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmu","Massive Multi-discipline Multimodal Understanding","multimodal","MMMU authors","2024",81.7,91.937,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmmupro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",70.6,24.5161,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmmupro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",70.6,24.5161,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mmmupro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",70.6,24.5161,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmmupro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.3,46.129,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmmupro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.3,46.129,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-mmmupro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.3,46.129,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-mmmupro-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",63,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-mmmupro-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",63,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-mmmupro-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",63,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mmmupro-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81,58.0645,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mmmupro-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81,58.0645,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mmmupro-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81,58.0645,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-mmmupro-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.9,67.4194,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-mmmupro-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.9,67.4194,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-mmmupro-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.9,67.4194,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mmmupro-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.6,66.4516,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mmmupro-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.6,66.4516,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mmmupro-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83.6,66.4516,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmupro-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",69.1,19.6774,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmupro-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",69.1,19.6774,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mmmupro-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",69.1,19.6774,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.8,34.8387,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.8,34.8387,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.8,34.8387,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmmupro-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.9,44.8387,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmmupro-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.9,44.8387,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-mmmupro-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.9,44.8387,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mmmupro-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.5,53.2258,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mmmupro-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.5,53.2258,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mmmupro-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.5,53.2258,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-mmmupro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",94,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-mmmupro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",94,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-mmmupro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",94,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mmmupro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mmmupro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mmmupro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupro-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.6,43.871,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupro-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.6,43.871,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupro-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",76.6,43.871,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupro-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",66.1,10,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupro-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",66.1,10,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupro-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",66.1,10,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupro-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupro-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupro-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.2,58.7097,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupro-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.4,49.6774,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupro-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.4,49.6774,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupro-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.4,49.6774,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupro-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83,64.5161,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupro-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83,64.5161,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupro-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",83,64.5161,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupro-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.7,57.0968,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupro-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.7,57.0968,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupro-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.7,57.0968,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-mmmupro-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.2,39.3548,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-mmmupro-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.2,39.3548,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-mmmupro-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.2,39.3548,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-mmmupro-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-mmmupro-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-mmmupro-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mmmupro-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.5,33.871,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mmmupro-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.5,33.871,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-mmmupro-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",73.5,33.871,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-mmmupro-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",74,35.4839,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmupro-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",71.1,26.129,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmupro-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",71.1,26.129,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-mmmupro-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",71.1,26.129,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmmupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmmupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-mmmupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmmupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmmupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmmupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupro-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.4,52.9032,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupro-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.4,52.9032,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupro-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79.4,52.9032,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupro-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.6,60,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupro-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.6,60,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupro-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",81.6,60,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmmupro-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.9,48.0645,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmmupro-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.9,48.0645,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-mmmupro-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",77.9,48.0645,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mmmupro-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mmmupro-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-mmmupro-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.1,48.7097,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-mmmupro-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.4,56.129,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-mmmupro-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.4,56.129,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-mmmupro-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",80.4,56.129,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmmupro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmmupro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mmmupro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmupro-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.8,41.2903,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmupro-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.8,41.2903,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mmmupro-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.8,41.2903,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmupro-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.8,50.9677,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmupro-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.8,50.9677,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mmmupro-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",78.8,50.9677,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.3,39.6774,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.3,39.6774,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",75.3,39.6774,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmupro-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmupro-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmmupro-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupro","Massive Multi-discipline Multimodal Understanding Pro","multimodal","MMMU-Pro authors","2024",79,51.6129,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mathvision-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",74.3,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mathvision-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",74.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-mathvision-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",74.3,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mathvision-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.6,61.5,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mathvision-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.6,61.5,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-mathvision-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.6,61.5,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mathvision-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",79.7,27,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mathvision-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",79.7,27,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mathvision-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",79.7,27,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mathvision-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83,43.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mathvision-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83,43.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-mathvision-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83,43.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mathvision-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",87.4,65.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mathvision-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",87.4,65.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mathvision-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",87.4,65.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mathvision-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",94.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mathvision-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",94.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mathvision-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86,58.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mathvision-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86,58.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mathvision-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86,58.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mathvision-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88.6,71.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mathvision-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88.6,71.5,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-mathvision-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88.6,71.5,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.2,59.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.2,59.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",86.2,59.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83.9,48,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83.9,48,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",83.9,48,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mathvision-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88,68.5,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mathvision-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88,68.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-mathvision-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",88,68.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mathvision-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",90.3,80,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mathvision-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",90.3,80,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mathvision-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvision","MathVision","multimodal","Qwen","2026",90.3,80,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mathvision-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-mathvision","MathVision","multimodal","Qwen","MathVision without tools",94.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports MathVision as 94.3 without tools and 97.8 with Python; this carrier is the no-tools track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:mathvision-with-ci:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-mathvision","MathVision","multimodal","Qwen","With CI",88.7,88.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:mathvision-with-ci:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-mathvision","MathVision","multimodal","Qwen","With CI",94.6,94.6,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-mathvision-with-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-mathvision","MathVision","multimodal","Qwen","With CI",94.6,94.6,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-mathvision:cell:vision:mathvision-with-ci:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-mathvision","MathVision","multimodal","Qwen","With CI",95.7,95.7,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:mathvision-no-ci:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",65.5,65.5,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-mathvision-without-ci-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",65.5,65.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-mathvision-without-ci-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",85.1,85.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:mathvision-no-ci:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",90.3,90.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-mathvision-without-ci-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",90.3,90.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:mathvision-no-ci:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",90,90,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-mathvision-without-ci","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",90,90,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-mathvision:cell:vision:mathvision-no-ci:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-mathvision","MathVision","multimodal","Qwen","Without CI",90.6,90.6,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","MathVision and CharXiv (RQ) report separate Without CI and With CI tracks; CI means code interpreter. Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-kimi-3-mathvisionpython-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvisionpython","MathVision with Python","multimodal","Moonshot AI / MathVision authors","2026",97.8,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mathvisionpython-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mathvisionpython","MathVision with Python","multimodal","Moonshot AI / MathVision authors","2026",97.8,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mathvisionpython-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-mathvisionpython","MathVision with Python","multimodal","Moonshot AI / MathVision authors","MathVision with Python",97.8,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports MathVision as 94.3 without tools and 97.8 with Python; this carrier is the with-Python track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-claude-opus-4-6-medxpertqamm-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",64.8,49.3865,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-medxpertqamm-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",64.8,49.3865,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-medxpertqamm-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",64.8,49.3865,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",81.3,100,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",81.3,100,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",81.3,100,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-medxpertqamm-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",48.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-medxpertqamm-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",48.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-medxpertqamm-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",48.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqamm-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",77.1,87.1166,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqamm-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",77.1,87.1166,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-medxpertqamm-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",77.1,87.1166,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqamm-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",65.8,52.454,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqamm-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",65.8,52.454,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-medxpertqamm-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",65.8,52.454,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqamm-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",78.4,91.1043,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqamm-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",78.4,91.1043,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-medxpertqamm-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",78.4,91.1043,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-medxpertqamm-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",71,68.4049,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-medxpertqamm-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",71,68.4049,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-medxpertqamm-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-medxpertqamm","MedXpertQA Multimodal","multimodal","Meta AI","2026",71,68.4049,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mlvuavg-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.6,33.3333,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mlvuavg-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.6,33.3333,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mlvuavg-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.6,33.3333,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",86.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mlvuavg-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",87.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mlvuavg-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",87.4,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mlvuavg-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mlvuavg","MLVU mean average","multimodal","Qwen","2026",87.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlongbenchdoc","MMLongBench-Doc","multimodal","Qwen","2026",57.5,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlongbenchdoc","MMLongBench-Doc","multimodal","Qwen","2026",57.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmlongbenchdoc","MMLongBench-Doc","multimodal","Qwen","2026",57.5,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-560","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu","MMMU","multimodal","MMMU authors",null,86,86,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-08-muse-glimmer-30b-mmmu-pro-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","1,730-question set",74,74,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["evidence-2026-08-muse-glimmer-30b-mmmu-pro-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","1,730-question set",74,74,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["mmmu-pro-owner-gemini-1-5-pro-0523","gemini-1-5-pro-may-2024","Gemini 1.5 Pro (May '24)","Gemini 1.5 Pro (0523)",null,"Gemini 1.5 Pro (0523)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2024",43.5,43.5,"percent","higher","2.2.0","ranking-eligible","direct","2024-05-23","2024-05-23","production::refresh-mmmu","refresh-mmmu","MMMU permanent refresh source","MMMU","https://mmmu-benchmark.github.io/","2024-05-23","2026-08-07","2026-08-18","official-leaderboard","Owner JSON has no source=author flag. Score matches the MMMU-Pro paper owner-evaluated table (overall 43.5; vision 40.5; standard 46.5). Not a provider self-report."],["mmmu-pro-owner-gpt-4o-0513","gpt-4o-2024-05-13","GPT-4o (May '24)","GPT-4o (0513)",null,"GPT-4o (0513)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2024",51.9,51.9,"percent","higher","2.2.0","ranking-eligible","direct","2024-05-13","2024-05-13","production::refresh-mmmu","refresh-mmmu","MMMU permanent refresh source","MMMU","https://mmmu-benchmark.github.io/","2024-05-13","2026-08-07","2026-08-18","official-leaderboard","Owner JSON has no source=author flag. Score matches the MMMU-Pro paper owner-evaluated table (overall 51.9; vision 49.7; standard 54.0). Not a provider self-report."],["mmmu-pro-owner-gpt-4o-mini","gpt-4o-mini","GPT-4o mini","GPT-4o mini",null,"GPT-4o mini","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2024",37.6,37.6,"percent","higher","2.2.0","ranking-eligible","direct","2024-07-18","2024-07-18","production::refresh-mmmu","refresh-mmmu","MMMU permanent refresh source","MMMU","https://mmmu-benchmark.github.io/","2024-07-18","2026-08-07","2026-08-18","official-leaderboard","Owner JSON has no source=author flag. Score matches the MMMU-Pro paper owner-evaluated table (overall 37.6; vision 35.2; standard 39.9). Not a provider self-report."],["benchlm-ref-claude-3-haiku-aammmupro-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",30.8,7.4394,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aammmupro-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",30.8,7.3883,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aammmupro-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",30.8,7.3883,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aammmupro-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.4,62.1107,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aammmupro-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.4,61.6838,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aammmupro-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.4,61.6838,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",67.9,71.6263,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",67.9,71.134,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",67.9,71.134,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammmupro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",71.2,77.3356,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammmupro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",71.2,76.8041,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aammmupro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",71.2,76.8041,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74,82.1799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74,81.6151,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74,81.6151,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aammmupro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.6021,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aammmupro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aammmupro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aammmupro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.5848,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aammmupro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aammmupro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aammmupro-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.8,90.4844,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aammmupro-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.8,89.8625,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aammmupro-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.8,89.8625,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aammmupro-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.4,86.3322,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aammmupro-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.4,85.7388,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aammmupro-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.4,85.7388,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aammmupro-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",84.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aammmupro-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",84.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aammmupro-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.6,76.2976,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aammmupro-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.6,75.7732,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aammmupro-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.6,75.7732,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aammmupro-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.8893,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aammmupro-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aammmupro-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aammmupro-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.2,63.4948,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aammmupro-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.2,63.0584,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aammmupro-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.2,63.0584,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aammmupro-2026-07-21","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55,49.308,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aammmupro-2026-07-27","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55,48.9691,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aammmupro-2026-08-01","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55,48.9691,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aammmupro-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.5,67.474,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aammmupro-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.5,67.0103,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aammmupro-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.5,67.0103,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aammmupro-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.9,83.737,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aammmupro-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.9,83.1615,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aammmupro-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.9,83.1615,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aammmupro-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,90.1384,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aammmupro-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aammmupro-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammmupro-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.2,92.9066,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammmupro-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.2,92.268,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aammmupro-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.2,92.268,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.7751,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.1924,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.1924,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aammmupro-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",82.4,96.7128,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aammmupro-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",82.4,96.0481,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aammmupro-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",82.4,96.0481,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aammmupro-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",84.3,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aammmupro-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",84.3,99.3127,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aammmupro-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",84.3,99.3127,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79,90.8304,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79,90.2062,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79,90.2062,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aammmupro-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.2,98.0969,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aammmupro-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.2,97.4227,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aammmupro-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.2,97.4227,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aammmupro-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48,37.1972,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aammmupro-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48,36.9416,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aammmupro-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48,36.9416,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aammmupro-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.7,74.7405,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aammmupro-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.7,74.2268,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aammmupro-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.7,74.2268,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.2,73.8754,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.2,73.3677,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.2,73.3677,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aammmupro-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.4,81.1419,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aammmupro-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.4,80.5842,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aammmupro-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.4,80.5842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aammmupro-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.6,31.3149,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aammmupro-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.6,31.0997,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aammmupro-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.6,31.0997,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aammmupro-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",51.4,43.0796,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aammmupro-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",51.4,42.7835,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aammmupro-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",51.4,42.7835,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aammmupro-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.8,80.1038,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aammmupro-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.8,79.5533,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aammmupro-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.8,79.5533,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aammmupro-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.2,60.0346,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aammmupro-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.2,59.622,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aammmupro-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.2,59.622,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aammmupro-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",58.7,55.7093,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aammmupro-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",58.7,55.3265,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aammmupro-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",58.7,55.3265,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aammmupro-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",40.1,23.5294,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aammmupro-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",40.1,23.3677,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aammmupro-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",40.1,23.3677,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aammmupro-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",41.5,25.9516,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aammmupro-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",41.5,25.7732,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aammmupro-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",41.5,25.7732,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,82.526,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,81.9588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,81.9588,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.699,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.1306,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.1306,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,82.526,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,81.9588,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aammmupro-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.2,81.9588,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.699,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.1306,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aammmupro-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.3,82.1306,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aammmupro-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.7751,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aammmupro-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.1924,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aammmupro-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.5,84.1924,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aammmupro-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.5848,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aammmupro-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aammmupro-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.5848,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.5,79.0378,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aammmupro-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.3,86.1592,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aammmupro-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.3,85.567,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aammmupro-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",76.3,85.567,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aammmupro-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.5,89.9654,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aammmupro-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.5,89.3471,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aammmupro-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.5,89.3471,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aammmupro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.4,89.7924,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aammmupro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.4,89.1753,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aammmupro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.4,89.1753,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aammmupro-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.3,80.9689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aammmupro-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.3,80.4124,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aammmupro-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.3,80.4124,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aammmupro-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.4,67.301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aammmupro-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.4,66.8385,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aammmupro-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",65.4,66.8385,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aammmupro-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.9,92.3875,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aammmupro-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.9,91.7526,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aammmupro-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.9,91.7526,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aammmupro-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,90.1384,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aammmupro-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aammmupro-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aammmupro-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.4,98.4429,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aammmupro-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.4,97.7663,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aammmupro-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",83.4,97.7663,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aammmupro-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.7,93.7716,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aammmupro-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.7,93.1271,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aammmupro-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.7,93.1271,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aammmupro-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",68.8,73.1834,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aammmupro-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",68.8,72.6804,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aammmupro-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",68.8,72.6804,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.8,61.0727,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.8,60.6529,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",61.8,60.6529,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aammmupro-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48.4,37.8893,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aammmupro-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48.4,37.6289,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aammmupro-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",48.4,37.6289,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.3,63.6678,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.3,63.2302,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",63.3,63.2302,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aammmupro-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.1,89.2734,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aammmupro-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.1,88.6598,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aammmupro-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.1,88.6598,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aammmupro-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.4,93.2526,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aammmupro-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.4,92.6117,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aammmupro-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.4,92.6117,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aammmupro-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.5,81.3149,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aammmupro-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.5,80.756,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aammmupro-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",73.5,80.756,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aammmupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.6021,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aammmupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aammmupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aammmupro-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.6021,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aammmupro-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aammmupro-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.4,84.0206,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aammmupro-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.4,91.5225,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aammmupro-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.4,90.8935,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aammmupro-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",79.4,90.8935,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aammmupro-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,93.4256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aammmupro-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aammmupro-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",26.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",26.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",26.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aammmupro-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.1,61.5917,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aammmupro-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.1,61.1684,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aammmupro-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",62.1,61.1684,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aammmupro-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",52.9,45.6747,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aammmupro-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",52.9,45.3608,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aammmupro-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",52.9,45.3608,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aammmupro-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.9,75.0865,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aammmupro-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.9,74.5704,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aammmupro-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",69.9,74.5704,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aammmupro-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,90.1384,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aammmupro-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aammmupro-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78.6,89.5189,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aammmupro-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55.7,50.519,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aammmupro-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55.7,50.1718,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aammmupro-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",55.7,50.1718,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aammmupro-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53,45.8478,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aammmupro-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53,45.5326,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aammmupro-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53,45.5326,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",64.9,66.436,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",64.9,65.9794,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",64.9,65.9794,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aammmupro-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.4221,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aammmupro-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.0619,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aammmupro-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.0619,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aammmupro-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.4221,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aammmupro-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.0619,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aammmupro-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",56.8,52.0619,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aammmupro-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,93.4256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aammmupro-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aammmupro-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53.2,46.1938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53.2,45.8763,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",53.2,45.8763,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aammmupro-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.3,30.7958,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aammmupro-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.3,30.5842,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aammmupro-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",44.3,30.5842,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aammmupro-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.1,75.4325,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aammmupro-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.1,74.9141,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aammmupro-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",70.1,74.9141,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aammmupro-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.91,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aammmupro-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aammmupro-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aammmupro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.8893,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aammmupro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aammmupro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aammmupro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.8893,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aammmupro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aammmupro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",77.3,87.2852,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.91,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.7,79.9308,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.7,79.3814,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",72.7,79.3814,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aammmupro-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.6,83.218,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aammmupro-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.6,82.646,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aammmupro-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",74.6,82.646,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aammmupro-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78,89.1003,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aammmupro-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78,88.488,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aammmupro-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",78,88.488,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.91,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75,83.3333,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aammmupro-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,93.4256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aammmupro-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aammmupro-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",80.5,92.7835,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aammmupro-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.3,84.4291,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aammmupro-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.3,83.8488,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aammmupro-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors","2026",75.3,83.8488,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-068","claude-haiku-4-5","Claude Haiku 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,55.14,55.14,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-claude-4-5-haiku-evals","aa-claude-4-5-haiku-evals","Artificial Analysis evaluations for claude-4-5-haiku","Artificial Analysis","https://artificialanalysis.ai/models/claude-4-5-haiku","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1394","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,67.9191,67.9191,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-639","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,70.6,70.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-637","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,77.3,77.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1727","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,61.7919,61.7919,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1381","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,68.7283,68.7283,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-064","claude-sonnet-5","Claude Sonnet 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,77.28,77.28,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-claude-sonnet-5-evals","aa-claude-sonnet-5-evals","Artificial Analysis evaluations for claude-sonnet-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-299","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,77.3,77.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-1635","command-a-plus","Command A+","Command A+",null,"Command A+","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,63.237,63.237,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1839","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,73.1214,73.1214,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1801","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.9133,74.9133,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-634","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81,81,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-300","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,75.5,75.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-072","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,82.43,82.43,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gemini-3-1-pro-preview-evals","aa-gemini-3-1-pro-preview-evals","Artificial Analysis evaluations for gemini-3-1-pro-preview","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-1-pro-preview","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-016","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.5,80.5,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-213","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.9,83.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-117","gemini-3-5-flash","Gemini 3.5 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.87,83.87,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gemini-3-5-flash-medium-evals","aa-gemini-3-5-flash-medium-evals","Artificial Analysis evaluations for gemini-3-5-flash-medium","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-medium","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-007","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.6,83.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-214","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.6,83.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2041","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis MMMU-Pro independent evaluation.",null,"Artificial Analysis MMMU-Pro independent evaluation.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,79,79,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA MMMU-Pro 79.0%."],["evidence-2026-07-2001","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis MMMU-Pro independent evaluation.",null,"Artificial Analysis MMMU-Pro independent evaluation.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.2,83.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA MMMU-Pro 83.2%."],["evidence-2026-07-1883","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,69.6532,69.6532,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1622","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,69.2486,69.2486,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-638","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,76.9,76.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1341","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.2197,74.2197,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1341--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.2197,74.2197,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1326","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,73.815,73.815,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1326--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,73.815,73.815,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1353","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,68.8439,68.8439,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1353--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,68.8439,68.8439,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1316","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,76.3006,76.3006,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1316--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,76.3006,76.3006,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-106","gpt-5-3-codex","GPT-5.3-Codex","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.5,78.5,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-3-codex-evals","aa-gpt-5-3-codex-evals","Artificial Analysis evaluations for gpt-5-3-codex","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-3-codex","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-298","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.5,78.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-102","gpt-5-4","GPT-5.4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.44,78.44,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-4-evals","aa-gpt-5-4-evals","Artificial Analysis evaluations for gpt-5-4","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-216","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81.2,81.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1295","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,73.2948,73.2948,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1295--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,73.2948,73.2948,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-640","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,66.1,66.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-113","gpt-5-5","GPT-5.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,79.88,79.88,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-5-evals","aa-gpt-5-5-evals","Artificial Analysis evaluations for gpt-5-5","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-026","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81.2,81.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-217","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81.2,81.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-054","gpt-5-6-luna","GPT-5.6 Luna","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.55,78.55,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-luna-evals","aa-gpt-5-6-luna-evals","Artificial Analysis evaluations for gpt-5-6-luna","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-221","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.4,78.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-046","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83.41,83.41,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-sol-evals","aa-gpt-5-6-sol-evals","Artificial Analysis evaluations for gpt-5-6-sol","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-215","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,83,83,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-050","gpt-5-6-terra","GPT-5.6 Terra","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.69,80.69,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-terra-evals","aa-gpt-5-6-terra-evals","Artificial Analysis evaluations for gpt-5-6-terra","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-218","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.7,80.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1660","grok-4","Grok 4","Grok 4",null,"Grok 4","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,68.8439,68.8439,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1764","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,61.7919,61.7919,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1449","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.5665,74.5665,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-076","grok-4-3","Grok 4.3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.09,78.09,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis evaluations for grok-4-3","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-223","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.1,78.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-120","grok-4-5","Grok 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.4,80.4,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-grok-4-5-evals","aa-grok-4-5-evals","Artificial Analysis evaluations for grok-4-5","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-297","grok-4-5","Grok 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.4,80.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-220","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.5,78.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1423","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,79.3642,79.3642,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1943","kimi-k3","Kimi K3","MMMU-Pro official protocol; original input order; images prepended to text; max reasoning; average of three runs.",null,"MMMU-Pro official protocol; original input order; images prepended to text; max reasoning; average of three runs.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81.6,81.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 81.6 on MMMU-Pro; 83.4 with Python also published."],["evidence-2026-07-1943--configuration--kimi-k3-max","kimi-k3","Kimi K3","MMMU-Pro official protocol; original input order; images prepended to text; max reasoning; average of three runs.","kimi-k3-max","MMMU-Pro official protocol; original input order; images prepended to text; max reasoning; average of three runs.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,81.6,81.6,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 81.6 on MMMU-Pro; 83.4 with Python also published."],["evidence-2026-07-1578","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,75.4335,75.4335,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-109","minimax-m3","MiniMax M3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.55,78.55,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-minimax-m3-evals","aa-minimax-m3-evals","Artificial Analysis evaluations for minimax-m3","Artificial Analysis","https://artificialanalysis.ai/models/minimax-m3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-222","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.1,78.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-094","mistral-large-3","Mistral Large 3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,55.66,55.66,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-mistral-large-3-evals","aa-mistral-large-3-evals","Artificial Analysis evaluations for mistral-large-3","Artificial Analysis","https://artificialanalysis.ai/models/mistral-large-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-302","mistral-large-3","Mistral Large 3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,55.7,55.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-086","mistral-medium-3-5","Mistral Medium 3.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,64.86,64.86,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-mistral-medium-3-5-evals","aa-mistral-medium-3-5-evals","Artificial Analysis evaluations for mistral-medium-3-5","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-090","mistral-small-4","Mistral Small 4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,56.82,56.82,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-mistral-small-4-evals","aa-mistral-small-4-evals","Artificial Analysis evaluations for mistral-small-4","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-301","mistral-small-4","Mistral Small 4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,56.8,56.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-635","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,80.4,80.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1648","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,64.5087,64.5087,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1366","o3","o3","o3",null,"o3","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,70.0578,70.0578,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1813","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,69.2486,69.2486,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1813--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,69.2486,69.2486,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1502","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.9711,74.9711,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1513","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,75.0289,75.0289,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1714","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,72.659,72.659,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1488","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,77.2832,77.2832,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1703","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,70.5202,70.5202,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1462","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,74.6243,74.6243,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-636","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,78.8,78.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-219","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,79,79,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-qwen-3-8-max-mmmu-pro","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,82.3,82.3,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 82.3 on MMMU-Pro."],["evidence-2026-08-qwen-3-8-max-mmmu-pro--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","mmmu-pro","MMMU-Pro","multimodal","MMMU-Pro authors",null,82.3,82.3,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 82.3 on MMMU-Pro."],["benchlm-ref-gpt-5-4-mmmupropython-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82.1,83.4437,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mmmupropython-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82.1,83.4437,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mmmupropython-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82.1,83.4437,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupropython-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",78,56.2914,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupropython-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",78,56.2914,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-mmmupropython-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",78,56.2914,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupropython-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",69.5,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupropython-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",69.5,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-mmmupropython-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",69.5,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupropython-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.2,90.7285,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupropython-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.2,90.7285,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mmmupropython-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.2,90.7285,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupropython-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",79.5,66.2252,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupropython-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",79.5,66.2252,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-mmmupropython-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",79.5,66.2252,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupropython-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",84.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupropython-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",84.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-mmmupropython-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",84.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupropython-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82,82.7815,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupropython-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82,82.7815,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-mmmupropython-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",82,82.7815,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupropython-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",80.1,70.1987,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupropython-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",80.1,70.1987,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-mmmupropython-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",80.1,70.1987,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupropython-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.4,92.053,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupropython-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.4,92.053,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-mmmupropython-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-mmmupropython","MMMU-Pro with Python","multimodal","OpenAI","2026",83.4,92.053,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmsearchplus-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmsearchplus","MMSearch-Plus","multimodal","Z.AI","2026",41.4,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmsearchplus-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmsearchplus","MMSearch-Plus","multimodal","Z.AI","2026",41.4,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mmsearchplus-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-mmsearchplus","MMSearch-Plus","multimodal","Z.AI","2026",41.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mstar-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mstar","MStar","multimodal","Qwen","2026",81.4,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mstar-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mstar","MStar","multimodal","Qwen","2026",81.4,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-mstar-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mstar","MStar","multimodal","Qwen","2026",81.4,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmvu-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",80.4,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmvu-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",80.4,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-mmvu-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",80.4,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmvu-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",73.3,12.3457,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmvu-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",73.3,12.3457,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-mmvu-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",73.3,12.3457,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",74.7,29.6296,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",74.7,29.6296,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",74.7,29.6296,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",72.3,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",72.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-mmvu","Multimodal Multi-disciplinary Video Understanding","multimodal","MMVU benchmark maintainers","2026",72.3,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-ocrbenchv2-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-ocrbenchv2-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-ocrbenchv2-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","ocrbench-v2","OCRBench v2","multimodal","OCRBench authors","2025",70.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",50.8,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",50.8,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",50.8,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-odinw13-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",51.1,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-odinw13-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",51.1,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-odinw13-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-odinw13","ODINW13","multimodal","Qwen","2026",51.1,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-officeqa-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-officeqa","OfficeQA","multimodal","Databricks and Anthropic","2026",78.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-officeqa-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-officeqa","OfficeQA","multimodal","Databricks and Anthropic","2026",78.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-olmocr-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-olmocr","olmOCR-Bench","multimodal","Allen Institute for AI","2025",85.7,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-olmocr-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-olmocr","olmOCR-Bench","multimodal","Allen Institute for AI","2025",85.7,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-olmocr-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-olmocr","olmOCR-Bench","multimodal","Allen Institute for AI","2025",85.7,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnidocbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench","OmniDocBench","multimodal","Moonshot AI / OmniDocBench authors","2026",91.1,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnidocbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench","OmniDocBench","multimodal","Moonshot AI / OmniDocBench authors","2026",91.1,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-omnidocbench-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-omnidocbench","OmniDocBench","multimodal","Moonshot AI / OmniDocBench authors","OmniDocBench",91.1,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports OmniDocBench=91.1 under its max-effort vision protocol. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["evidence-2026-08-15-claude-opus-4-6-benchlm-omnidocbench15-1-5-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","1.5",86.6,86.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-benchlm-omnidocbench15-1-5-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","1.5",89.4,89.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-omnidocbench15-1-5-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","1.5",91.4,91.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-omnidocbench15-1-5","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","1.5",91.1,91.1,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-minimax-m3-omnidocbench15-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.6,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omnidocbench15-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.6,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-omnidocbench15-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.6,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",89.9,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",89.9,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",89.9,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnidocbench15-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.4,88.2353,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnidocbench15-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.4,88.2353,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-omnidocbench15-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","2026",91.4,88.2353,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-benchlm-omnidocbench15-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","v1.5 / 1,355 prompts",75.8,75.8,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only because Meta documents a modified two-component scoring and matching implementation."],["evidence-2026-08-muse-glimmer-30b-benchlm-omnidocbench15-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","benchlm-omnidocbench15","OmniDocBench 1.5","multimodal","OpenAI","v1.5 / 1,355 prompts",75.8,75.8,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only because Meta documents a modified two-component scoring and matching implementation."],["benchlm-ref-kimi-3-perceptionbench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-perceptionbench","PerceptionBench (Internal)","multimodal","Moonshot AI","2026",58.5,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-perceptionbench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-perceptionbench","PerceptionBench (Internal)","multimodal","Moonshot AI","2026",58.5,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-perceptionbench-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-perceptionbench","PerceptionBench (Internal)","multimodal","Moonshot AI","PerceptionBench Kimi K3 launch evaluation",58.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports PerceptionBench=58.5 and identifies it as an in-house atomic visual-perception benchmark. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:realworldqa:3","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",73.9,73.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-benchlm-realworldqa-2026-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",73.9,73.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-07-21","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",58.43,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-07-27","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",58.43,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-08-01","lfm2-5-vl-450m","LFM2.5-VL-450M","Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",58.43,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-realworldqa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",84.1,90.1651,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-realworldqa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",84.1,90.1651,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-realworldqa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",84.1,90.1651,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-realworldqa-2026-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",84.1,84.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",85.3,94.38,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",85.3,94.38,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",85.3,94.38,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-realworldqa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",86.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-realworldqa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",86.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-realworldqa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",86.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:realworldqa:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",86.9,86.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-realworldqa-2026-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",86.9,86.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:realworldqa:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",85.9,85.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-realworldqa-2026","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",85.9,85.9,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:benchlm-realworldqa:cell:vision:realworldqa:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","benchlm-realworldqa","RealWorldQA","multimodal","Qwen","2026",88.5,88.5,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["benchlm-ref-interfaze-beta-refcocoavg-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",82.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-refcocoavg-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",82.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-refcocoavg-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",82.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",90.5,80.7692,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",90.5,80.7692,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",90.5,80.7692,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-refcocoavg-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-refcocoavg-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-refcocoavg-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92,95.1923,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92,95.1923,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-refcocoavg","RefCOCO average","multimodal","RefCOCO dataset authors","2026",92,95.1923,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-screenspotpro-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",45.7,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-screenspotpro-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",45.7,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-screenspotpro-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",45.7,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-screenspotpro-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",83.1,88.6256,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-screenspotpro-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",83.1,88.6256,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-screenspotpro-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",83.1,88.6256,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-screenspotpro-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",87.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-screenspotpro-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",87.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-screenspotpro-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",87.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-screenspotpro-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",72.7,63.981,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-screenspotpro-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",72.7,63.981,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-screenspotpro-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",72.7,63.981,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-screenspotpro-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.4,91.7062,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-screenspotpro-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.4,91.7062,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-screenspotpro-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.4,91.7062,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-screenspotpro-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",85.4,94.0758,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-screenspotpro-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",85.4,94.0758,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-screenspotpro-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",85.4,94.0758,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-235b-a22b-screenspotpro-2026-07-21","holo2-235b-a22b","Holo2-235B-A22B","Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",70.6,59.0047,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-235b-a22b-screenspotpro-2026-07-27","holo2-235b-a22b","Holo2-235B-A22B","Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",70.6,59.0047,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-235b-a22b-screenspotpro-2026-08-01","holo2-235b-a22b","Holo2-235B-A22B","Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",70.6,59.0047,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-30b-a3b-screenspotpro-2026-07-21","holo2-30b-a3b","Holo2-30B-A3B","Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",66.1,48.3412,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-30b-a3b-screenspotpro-2026-07-27","holo2-30b-a3b","Holo2-30B-A3B","Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",66.1,48.3412,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-30b-a3b-screenspotpro-2026-08-01","holo2-30b-a3b","Holo2-30B-A3B","Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",66.1,48.3412,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-4b-screenspotpro-2026-07-21","holo2-4b","Holo2-4B","Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.2,27.2512,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-4b-screenspotpro-2026-07-27","holo2-4b","Holo2-4B","Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.2,27.2512,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-4b-screenspotpro-2026-08-01","holo2-4b","Holo2-4B","Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.2,27.2512,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-8b-screenspotpro-2026-07-21","holo2-8b","Holo2-8B","Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",58.9,31.2796,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-8b-screenspotpro-2026-07-27","holo2-8b","Holo2-8B","Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",58.9,31.2796,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-holo2-8b-screenspotpro-2026-08-01","holo2-8b","Holo2-8B","Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",58.9,31.2796,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-screenspotpro-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.1,90.9953,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-screenspotpro-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.1,90.9953,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-screenspotpro-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",84.1,90.9953,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.8,28.673,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.8,28.673,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",57.8,28.673,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-screenspotpro-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",65.6,47.1564,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-screenspotpro-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",65.6,47.1564,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-screenspotpro-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",65.6,47.1564,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-screenspotpro-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",68.2,53.3175,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-screenspotpro-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",68.2,53.3175,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-screenspotpro-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",68.2,53.3175,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-screenspotpro-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",79,78.91,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-screenspotpro-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",79,78.91,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-screenspotpro-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","2025",79,78.91,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-muse-glimmer-30b-screenspot-pro-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","Pro / 1,581 examples",75.4,75.4,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only point-in-box accuracy under Meta's documented crop-and-verify implementation."],["evidence-2026-08-muse-glimmer-30b-screenspot-pro-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","screenspot-pro","ScreenSpot-Pro","multimodal","ScreenSpot","Pro / 1,581 examples",75.4,75.4,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Reference-only point-in-box accuracy under Meta's documented crop-and-verify implementation."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-simplevqa:simplevqa:2","claude-opus-5","Claude Opus 5","Claude Opus 5 (max) as published in Meta's comparison table","claude-opus-5-max","Claude Opus 5 (max) as published in Meta's comparison table","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2024 English and Chinese set",70.3,70.3,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. The 2024 set contains 1,013 English and 1,011 Chinese questions; raw accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-simplevqa:simplevqa:4","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high) as published in Meta's comparison table","gemini-3-7-flash-high","Gemini 3.7 Flash (high) as published in Meta's comparison table","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2024 English and Chinese set",77.8,77.8,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. The 2024 set contains 1,013 English and 1,011 Chinese questions; raw accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-simplevqa:simplevqa:1","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max) as published in Meta's comparison table","gpt-5-6-sol-max","GPT-5.6 Sol (max) as published in Meta's comparison table","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2024 English and Chinese set",64.7,64.7,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. The 2024 set contains 1,013 English and 1,011 Chinese questions; raw accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-simplevqa:simplevqa:3","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2024 English and Chinese set",73.3,73.3,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. The 2024 set contains 1,013 English and 1,011 Chinese questions; raw accuracy is reported with GPT-OSS-120B high as judge. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["benchlm-ref-gemini-3-1-pro-simplevqa-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",72.4,63.6719,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-simplevqa-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",72.4,63.6719,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-simplevqa-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",72.4,63.6719,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-simplevqa-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",61.1,19.5313,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-simplevqa-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",61.1,19.5313,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-simplevqa-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",61.1,19.5313,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-simplevqa-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",57.4,5.0781,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-simplevqa-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",57.4,5.0781,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-simplevqa-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",57.4,5.0781,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-simplevqa-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",71.3,59.375,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-simplevqa-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",71.3,59.375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-simplevqa-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",71.3,59.375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-simplevqa-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",56.1,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-simplevqa-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",56.1,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-simplevqa-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",56.1,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",58.9,10.9375,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",58.9,10.9375,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",58.9,10.9375,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-simplevqa-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",81.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-simplevqa-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",81.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-simplevqa-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",81.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-simplevqa-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",79.2,90.2344,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-simplevqa-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",79.2,90.2344,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-simplevqa-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-simplevqa","SimpleVQA","multimodal","Z.AI","2026",79.2,90.2344,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vstar-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",67,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vstar-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",67,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-vstar-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",67,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vstar-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",88,70.2341,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vstar-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",88,70.2341,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-vstar-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",88,70.2341,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vstar-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",75.9,29.7659,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vstar-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",75.9,29.7659,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-vstar-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",75.9,29.7659,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vstar-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vstar-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-vstar-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-vstar-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.7,89.2977,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-vstar-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.7,89.2977,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-vstar-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.7,89.2977,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vstar-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.8,96.3211,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vstar-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.8,96.3211,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-vstar-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.8,96.3211,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-vstar-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.2,87.6254,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-vstar-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.2,87.6254,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-vstar-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",93.2,87.6254,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-vstar-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",92.7,85.9532,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-vstar-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",92.7,85.9532,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-vstar-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",92.7,85.9532,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-vstar-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",94.7,92.6421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-vstar-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",94.7,92.6421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-vstar-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",94.7,92.6421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vstar-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vstar-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-vstar-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",96.9,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-vstar-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.3,94.6488,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-vstar-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.3,94.6488,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-vstar-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-vstar","V*","multimodal","Z.AI","2026",95.3,94.6488,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videomme-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videomme","Video-MME","multimodal","Video-MME benchmark team","2024",87.4,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videomme-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videomme","Video-MME","multimodal","Video-MME benchmark team","2024",87.4,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videomme-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videomme","Video-MME","multimodal","Video-MME benchmark team","2024",87.4,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-videommewithsub-2026-07-21","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-videommewithsub-2026-07-27","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-videommewithsub-2026-08-01","mimo-v2-5","MiMo-V2.5","Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommewithsub-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",85.4,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommewithsub-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",85.4,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommewithsub-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",85.4,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommewithsub-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommewithsub-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommewithsub-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",87.7,88.4615,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",86.6,46.1538,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",86.6,46.1538,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",86.6,46.1538,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommewithsub-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",88,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommewithsub-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",88,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommewithsub-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-videommewithsub","Video-MME with subtitle","multimodal","Qwen","2026",88,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",72.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",72.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",72.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",82.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",82.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-videommenosub","Video-MME without subtitle","multimodal","Qwen","2026",82.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-videommmu-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-videommmu-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-videommmu-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-videommmu-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",87.6,100,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-videommmu-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",87.6,100,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-videommmu-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",87.6,100,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videommmu-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",86.6,74.359,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videommmu-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",86.6,74.359,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-videommmu-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",86.6,74.359,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommmu-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.6,23.0769,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommmu-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.6,23.0769,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-videommmu-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.6,23.0769,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-videommmu-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.7,25.641,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-videommmu-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.7,25.641,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-videommmu-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.7,25.641,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommmu-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommmu-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-videommmu-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84.4,17.9487,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-videommmu-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84,7.6923,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-videommmu-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84,7.6923,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-videommmu-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",84,7.6923,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",83.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",83.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",83.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommmu-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",85.4,43.5897,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommmu-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",85.4,43.5897,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-videommmu-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","video-mmmu","Video-MMMU","multimodal","Video-MMMU","2026",85.4,43.5897,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-qwen3-6-27b-benchlm-vision2web-frontend-webpage-website-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","benchlm-vision2web","Vision2Web","multimodal","Z.AI","frontend/webpage/website",45,45,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:vision2web:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-vision2web","Vision2Web","multimodal","Z.AI","frontend/webpage/website",42.1,42.1,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Vision2Web was run with Claude Code and judged by gpt-5.4-2026-03-05. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-benchlm-vision2web-frontend-webpage-website-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","benchlm-vision2web","Vision2Web","multimodal","Z.AI","frontend/webpage/website",42.1,42.1,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:vision:vision2web:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","benchlm-vision2web","Vision2Web","multimodal","Z.AI","frontend/webpage/website",62.9,62.9,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Vision2Web was run with Claude Code and judged by gpt-5.4-2026-03-05. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-benchlm-vision2web-frontend-webpage-website","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","benchlm-vision2web","Vision2Web","multimodal","Z.AI","frontend/webpage/website",62.9,62.9,"percent","higher","2.1.0","reference-only","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["benchlm-ref-interfaze-beta-voxpopuliwer-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-voxpopuliwer","VoxPopuli-Cleaned-AA Word Error Rate","multimodal","Artificial Analysis / VoxPopuli dataset authors","2026",2.4,50,"percent","lower","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-voxpopuliwer-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-voxpopuliwer","VoxPopuli-Cleaned-AA Word Error Rate","multimodal","Artificial Analysis / VoxPopuli dataset authors","2026",2.4,50,"percent","lower","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-voxpopuliwer-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","benchlm-voxpopuliwer","VoxPopuli-Cleaned-AA Word Error Rate","multimodal","Artificial Analysis / VoxPopuli dataset authors","2026",2.4,50,"percent","lower","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-worldvqaforceanswer-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-worldvqaforceanswer","WorldVQA ForceAnswer","multimodal","Moonshot AI / WorldVQA authors","2026",51,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-worldvqaforceanswer-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-worldvqaforceanswer","WorldVQA ForceAnswer","multimodal","Moonshot AI / WorldVQA authors","2026",51,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-worldvqaforceanswer-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-worldvqaforceanswer","WorldVQA ForceAnswer","multimodal","Moonshot AI / WorldVQA authors","WorldVQA ForceAnswer",51,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports WorldVQA ForceAnswer=51.0 under its max-effort vision protocol. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-gemini-3-1-pro-zerobench-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",29,33.3333,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-zerobench-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",29,33.3333,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-zerobench-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",29,33.3333,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-zerobench-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",41,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-zerobench-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",41,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-zerobench-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",41,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-zerobench-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",23,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-zerobench-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",23,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-zerobench-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",33,55.5556,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-zerobench-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",33,55.5556,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-zerobench-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","zerobench","ZeroBench","multimodal","ZeroBench","2026",33,55.5556,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["deepseek-v4-flash-vision-exp-zerobench-pass-at-5-2026-08-21","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","Provider chart specifies Pass@5 but not sampling or tool settings",null,"Provider chart specifies Pass@5 but not sampling or tool settings","zerobench","ZeroBench","multimodal","ZeroBench","August 2026",35,35,"percent","higher","2.3.0","reference-only","direct","2026-08-21","2026-08-21","production::deepseek-v4-flash-vision-exp-release","deepseek-v4-flash-vision-exp-release","DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","DeepSeek","https://api-docs.deepseek.com/zh-cn/updates/","2026-08-21","2026-08-27","2026-08-24","provider-reported","Official DeepSeek provider subject value reported as Pass@5; exact sampling, harness, tool policy and configuration are unavailable in the chart."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-zerobench:zerobench:2","claude-opus-5","Claude Opus 5","Claude Opus 5 (max) as published in Meta's comparison table","claude-opus-5-max","Claude Opus 5 (max) as published in Meta's comparison table","zerobench","ZeroBench","multimodal","ZeroBench","rolling",47.5,47.5,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 100 stochastic visual-reasoning questions; GPT-OSS-120B high with a container acted as judge; Pass@5 is reported. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-zerobench:zerobench:4","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high) as published in Meta's comparison table","gemini-3-7-flash-high","Gemini 3.7 Flash (high) as published in Meta's comparison table","zerobench","ZeroBench","multimodal","ZeroBench","rolling",30,30,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 100 stochastic visual-reasoning questions; GPT-OSS-120B high with a container acted as judge; Pass@5 is reported. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-zerobench:zerobench:1","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max) as published in Meta's comparison table","gpt-5-6-sol-max","GPT-5.6 Sol (max) as published in Meta's comparison table","zerobench","ZeroBench","multimodal","ZeroBench","rolling",54.6,54.6,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 100 stochastic visual-reasoning questions; GPT-OSS-120B high with a container acted as judge; Pass@5 is reported. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["provided-comparison:muse-spark-1-2-multimodal-2026-08-27:cell:muse-spark-1-2-zerobench:zerobench:3","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh) as published in Meta's comparison table","zerobench","ZeroBench","multimodal","ZeroBench","rolling",46,46,"percent","higher","2.1.0","excluded","direct","2026-08-20","2026-08-20","production::meta-muse-spark-1-2-multimodal-2026-08-20","meta-muse-spark-1-2-multimodal-2026-08-20","Multimodal Intelligence of Muse Spark 1.2","Meta Superintelligence Labs","https://research.meta.ai/blog/multimodal-intelligence-of-muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","provider-reported","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90. 100 stochastic visual-reasoning questions; GPT-OSS-120B high with a container acted as judge; Pass@5 is reported. Competitor cell retained from the complete official provider table; it is never independent direct evidence."],["new-model:aa-muse-spark-1-2-2026-08-27:muse-spark-1-2:zerobench:cell:muse-spark-1-2-zerobench:zerobench:0","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","zerobench","ZeroBench","multimodal","ZeroBench","rolling",54,54,"percent","higher","2.1.0","reference-only","direct","2026-08-20",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-20","2026-08-27","2026-08-27","source-checked","Models ran with their documented effort settings in isolated containers with headless coding and GUI tools; internet access was blocked. Images were limited to a 2,000-pixel maximum side and saved with Pillow quality 90.; 100 stochastic visual-reasoning questions; GPT-OSS-120B high with a container acted as judge; Pass@5 is reported.; pass@5"],["benchlm-ref-kimi-3-zerobench-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","zerobench","ZeroBench","multimodal","ZeroBench","ZeroBench pass@5 without tools",23,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports ZeroBench pass@5 as 23.0 without tools and 41.0 with Python; this carrier is the no-tools track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["benchlm-ref-kimi-3-zerobenchpython-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-zerobenchpython","ZeroBench_main with Python","multimodal","Moonshot AI / ZeroBench authors","2026",41,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-zerobenchpython-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","benchlm-zerobenchpython","ZeroBench_main with Python","multimodal","Moonshot AI / ZeroBench authors","2026",41,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-zerobenchpython-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","benchlm-zerobenchpython","ZeroBench_main with Python","multimodal","Moonshot AI / ZeroBench authors","ZeroBench pass@5 with Python",41,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports ZeroBench pass@5 as 23.0 without tools and 41.0 with Python; this carrier is the with-Python track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["evidence-2026-08-muse-glimmer-30b-aa-lcr-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","100-question set",80,80,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Exact no-tools High-reasoning result sourced through Meta's methodology."],["evidence-2026-08-muse-glimmer-30b-aa-lcr-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","100-question set",80,80,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Exact no-tools High-reasoning result sourced through Meta's methodology."],["benchlm-ref-claude-3-haiku-lcr-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",21,27.7411,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-lcr-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",21,27.7411,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-lcr-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",21,27.7411,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-lcr-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.3,58.5205,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-lcr-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.3,58.5205,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-lcr-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.3,58.5205,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-lcr-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-lcr-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-lcr-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-lcr-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70,92.4703,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-lcr-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70,92.4703,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-lcr-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70,92.4703,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-lcr-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-lcr-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-lcr-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-lcr-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-lcr-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-lcr-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-lcr-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-lcr-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-lcr-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-lcr-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",58.3,77.0145,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-lcr-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",58.3,77.0145,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-lcr-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",58.3,77.0145,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-lcr-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.3,92.8666,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-lcr-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.3,92.8666,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-lcr-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.3,92.8666,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-lcr-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-lcr-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-lcr-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-lcr-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-lcr-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-lcr-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-lcr-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70,92.4703,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-lcr-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70,92.4703,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-lcr-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",57.7,76.2219,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-lcr-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",57.7,76.2219,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-lcr-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",57.7,76.2219,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-lcr-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-lcr-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-lcr-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-lcr-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-lcr-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-lcr-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-07-21","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",9.7,12.8137,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-07-27","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",9.7,12.8137,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-08-01","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",9.7,12.8137,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-lcr-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",29,38.3091,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-lcr-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",29,38.3091,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-lcr-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",29,38.3091,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-lcr-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",53.3,70.4095,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-lcr-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",53.3,70.4095,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-lcr-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",53.3,70.4095,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-lcr-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45,59.4452,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-lcr-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45,59.4452,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-lcr-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45,59.4452,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-lcr-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",39,51.5192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-lcr-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",39,51.5192,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-lcr-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",39,51.5192,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-lcr-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-lcr-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63,83.2232,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-lcr-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-lcr-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-lcr-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",54.7,72.2589,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-lcr-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",54.7,72.2589,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-lcr-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",54.7,72.2589,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-lcr-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-lcr-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-lcr-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-lcr-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",8,10.568,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-lcr-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",8,10.568,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-lcr-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",8,10.568,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-lcr-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45.9,60.6341,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-lcr-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45.9,60.6341,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-lcr-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",45.9,60.6341,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-lcr-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-lcr-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-lcr-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-lcr-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48,63.4082,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-lcr-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48,63.4082,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-lcr-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48,63.4082,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-lcr-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-lcr-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-lcr-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",70.7,93.395,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-lcr-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-lcr-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-lcr-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-lcr-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-lcr-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-lcr-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lcr-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lcr-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lcr-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-lcr-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-lcr-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-lcr-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-lcr-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-lcr-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-lcr-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-lcr-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.7,7.5297,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-lcr-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.7,7.5297,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-lcr-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.7,7.5297,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-lcr-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.3,73.0515,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-lcr-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.3,73.0515,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-lcr-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.3,73.0515,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-lcr-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-lcr-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-lcr-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-lcr-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-lcr-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-lcr-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62,81.9022,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-lcr-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",15,19.8151,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-lcr-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",15,19.8151,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-lcr-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",15,19.8151,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-lcr-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-lcr-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-lcr-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-lcr-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",43.7,57.7279,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-lcr-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",43.7,57.7279,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-lcr-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",43.7,57.7279,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-lcr-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",26.3,34.7424,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-lcr-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",26.3,34.7424,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-lcr-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",26.3,34.7424,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-lcr-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64,84.5443,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-lcr-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64,84.5443,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-lcr-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64,84.5443,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-lcr-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-lcr-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-lcr-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-lcr-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-lcr-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-lcr-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-lcr-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.3,82.2985,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-lcr-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.3,82.2985,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-lcr-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.3,82.2985,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-lcr-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",71.3,94.1876,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-lcr-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",71.3,94.1876,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-lcr-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",71.3,94.1876,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-lcr-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-lcr-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-lcr-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-lcr-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-lcr-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-lcr-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-lcr-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",42.3,55.8785,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-lcr-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",42.3,55.8785,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-lcr-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",42.3,55.8785,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-lcr-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",17,22.4571,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-lcr-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",17,22.4571,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-lcr-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",17,22.4571,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-lcr-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-lcr-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-lcr-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-lcr-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.6,99.8679,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-lcr-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.8,96.1691,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-lcr-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75,99.0753,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-lcr-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75,99.0753,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-lcr-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75,99.0753,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-lcr-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-lcr-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-lcr-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-lcr-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-lcr-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-lcr-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-lcr-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-lcr-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-lcr-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",72.7,96.037,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-lcr-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.7,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-lcr-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.7,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-lcr-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",75.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-lcr-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-lcr-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-lcr-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-lcr-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-lcr-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-lcr-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-lcr-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-lcr-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-lcr-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-lcr-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-lcr-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-lcr-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66,87.1863,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-lcr-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.3,98.1506,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-lcr-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.3,98.1506,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-lcr-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.3,98.1506,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-lcr-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-lcr-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-lcr-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-lcr-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.7,97.358,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-lcr-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.7,97.358,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-lcr-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.7,97.358,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-lcr-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-lcr-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-lcr-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-lcr-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",50.7,66.9749,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-lcr-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",50.7,66.9749,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-lcr-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",50.7,66.9749,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-lcr-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-lcr-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-lcr-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",30.7,40.5548,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-lcr-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",4,5.284,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-lcr-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",4,5.284,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-lcr-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",4,5.284,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-lcr-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-lcr-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-lcr-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-lcr-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",6.3,8.3223,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-lcr-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",6.3,8.3223,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-lcr-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",6.3,8.3223,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-lcr-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-lcr-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-lcr-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-lcr-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-lcr-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-lcr-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-lcr-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.7,85.469,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-lcr-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.7,85.469,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-lcr-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.7,85.469,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-lcr-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",22,29.0621,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-lcr-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",22,29.0621,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-lcr-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",22,29.0621,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68,89.8283,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-lcr-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.3,84.9406,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-lcr-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.3,84.9406,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-lcr-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",64.3,84.9406,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-lcr-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-lcr-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-lcr-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.7,89.432,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-lcr-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48.3,63.8045,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-lcr-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48.3,63.8045,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-lcr-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",48.3,63.8045,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-lcr-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-lcr-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-lcr-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-lcr-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-lcr-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-lcr-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-lcr-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-lcr-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-lcr-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-lcr-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-lcr-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-lcr-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",55.7,73.5799,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-lcr-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",51,67.3712,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-lcr-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",51,67.3712,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-lcr-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",51,67.3712,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-lcr-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-lcr-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-lcr-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-lcr-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-lcr-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-lcr-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.3,86.2616,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-lcr-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-lcr-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-lcr-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-lcr-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-lcr-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-lcr-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.3,87.5826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lcr-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.7,98.679,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lcr-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.7,98.679,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-lcr-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74.7,98.679,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-lcr-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-lcr-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-lcr-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-lcr-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25,33.0251,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-lcr-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25,33.0251,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-lcr-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25,33.0251,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-lcr-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",24.3,32.1004,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-lcr-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",24.3,32.1004,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-lcr-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",24.3,32.1004,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-lcr-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-lcr-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-lcr-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46,60.7662,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-lcr-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25.8,34.0819,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-lcr-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25.8,34.0819,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-lcr-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",25.8,34.0819,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-lcr-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",31.3,41.3474,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-lcr-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",31.3,41.3474,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-lcr-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",31.3,41.3474,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-lcr-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-lcr-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-lcr-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-lcr-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-lcr-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-lcr-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",60.7,80.1849,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-lcr-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.3,96.8296,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-lcr-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.3,96.8296,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-lcr-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",73.3,96.8296,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-lcr-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-lcr-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-lcr-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-lcr-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-lcr-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-lcr-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",74,97.7543,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-lcr-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.3,7.0013,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-lcr-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.3,7.0013,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-lcr-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",5.3,7.0013,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-lcr-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",34.7,45.8388,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-lcr-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",34.7,45.8388,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-lcr-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",34.7,45.8388,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-lcr-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",28,36.9881,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-lcr-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",28,36.9881,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-lcr-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",28,36.9881,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-lcr-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-lcr-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-lcr-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",61,80.5812,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-lcr-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-lcr-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-lcr-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-lcr-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-lcr-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-lcr-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",44.7,59.0489,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-lcr-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-lcr-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-lcr-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-lcr-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-lcr-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-lcr-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.3,83.6196,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-lcr-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33.7,44.5178,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-lcr-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33.7,44.5178,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-lcr-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33.7,44.5178,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",35.7,47.1598,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",35.7,47.1598,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",35.7,47.1598,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-lcr-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-lcr-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-lcr-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67,88.5073,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-lcr-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",7.3,9.6433,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-lcr-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",7.3,9.6433,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-lcr-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",7.3,9.6433,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-lcr-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",19,25.0991,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-lcr-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",19,25.0991,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-lcr-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",19,25.0991,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-lcr-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",59.3,78.3355,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-lcr-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",59.3,78.3355,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-lcr-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",59.3,78.3355,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-lcr-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-lcr-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-lcr-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.3,91.5456,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-lcr-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-lcr-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-lcr-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-lcr-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-lcr-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-lcr-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-lcr-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46.7,61.6909,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-lcr-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46.7,61.6909,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-lcr-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",46.7,61.6909,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-lcr-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-lcr-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-lcr-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",67.3,88.9036,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-lcr-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-lcr-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-lcr-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-lcr-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-lcr-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-lcr-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65.7,86.79,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-lcr-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-lcr-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-lcr-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",66.7,88.111,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-lcr-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-lcr-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-lcr-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",62.7,82.8269,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-lcr-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-lcr-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-lcr-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",68.7,90.753,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-lcr-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-lcr-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-lcr-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69.7,92.074,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-lcr-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-lcr-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-lcr-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-lcr-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69,91.1493,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-lcr-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69,91.1493,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-lcr-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",69,91.1493,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-lcr-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-lcr-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-lcr-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",65,85.8653,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-lcr-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-lcr-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-lcr-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-lcr-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-lcr-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-lcr-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-lcr-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-lcr-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-lcr-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-lcr-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-lcr-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-lcr-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",63.7,84.148,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-lcr-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-lcr-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-lcr-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-lcr-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-lcr-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-lcr-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","2026",33,43.5931,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["aa-individual:claude-fable-5:aa-lcr:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",76.6666666666667,76.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:claude-opus-4-6-thinking:aa-lcr:2026-08-29","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",74.333333333333,74.333333333333,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-6-thinking-2026-08-29","aa-current-claude-opus-4-6-thinking-2026-08-29","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-6-adaptive","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:claude-opus-4-7-adaptive:aa-lcr:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",75.333333333333,75.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:claude-opus-4-8:aa-lcr:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",73,73,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:aa-lcr:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",75.6666666666667,75.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-sonnet-5:aa-lcr:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",77,77,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["evidence-2026-08-15-command-a-plus-aa-lcr-standard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",46,46,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["aa-current:deepseek-v3-1:aa-lcr:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","deepseek-v3-1-non-reasoning","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",46.666666666667,46.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-2026-08-29","aa-current-deepseek-v3-1-2026-08-29","DeepSeek V3.1 (Non-reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:deepseek-v3-1-reasoning:aa-lcr:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","deepseek-v3-1-reasoning-default","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",56.666666666667,56.666666666667,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-reasoning-2026-08-29","aa-current-deepseek-v3-1-reasoning-2026-08-29","DeepSeek V3.1 (Reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["evidence-2026-08-15-deepseek-v4-flash-aa-lcr-standard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",63.7,63.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["aa-individual:deepseek-v4-flash-vision-exp:aa-lcr:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",78,78,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual aa-lcr result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-individual:deepseek-v4-pro-0813:aa-lcr:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",75.3333333333333,75.3333333333333,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-6-flash:aa-lcr:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",79,79,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:aa-lcr:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",81,81,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:aa-lcr:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",6.333333333333,6.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gemma-4-26b-a4b:aa-lcr:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",61.666666666667,61.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:glm-5-2:aa-lcr:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",76.666666666667,76.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:aa-lcr:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",76.3333333333333,76.3333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:aa-lcr:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",78,78,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual aa-lcr result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:aa-lcr:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",45.333333333333,45.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:aa-lcr:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",19.333333333333,19.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:gpt-5-4:aa-lcr:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",77.6666666666667,77.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-5:aa-lcr:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",79,79,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-luna:aa-lcr:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",78.3333333333333,78.3333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-sol:aa-lcr:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",77.6666666666667,77.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-terra:aa-lcr:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",79.6666666666667,79.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-5:aa-lcr:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",74,74,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-6:aa-lcr:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",75,75,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:kimi-k2-5-reasoning:aa-lcr:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",73,73,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:kimi-k3:aa-lcr:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",82.6666666666667,82.6666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:llama-4-maverick:aa-lcr:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",50,50,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:aa-lcr:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",30.333333333333,30.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:aa-lcr:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",62.666666666667,62.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["evidence-2026-08-15-mimo-v2-5-aa-lcr-standard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",62.7,62.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-aa-lcr-standard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",61,61,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["aa-current:mistral-medium-3-5-128b:aa-lcr:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",65.333333333333,65.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:mistral-small-4-reasoning:aa-lcr:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",47.333333333333,47.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:muse-spark-1-1:aa-lcr:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",81.3333333333333,81.3333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:muse-spark-1-2:aa-lcr:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",83.3333333333333,83.3333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual aa-lcr result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:aa-lcr:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",37.333333333333,37.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-397b-reasoning:aa-lcr:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",72.666666666667,72.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:aa-lcr:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",70.333333333333,70.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-6-35b-a3b:aa-lcr:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",66.666666666667,66.666666666667,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen-3-8-flash-next:aa-lcr:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",77,77,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["evidence-2026-08-15-solar-open-100b-reasoning-aa-lcr-standard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",36,36,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-aa-lcr-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",62.3,62.3,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["aa-current:trinity-large-thinking:aa-lcr:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis","standard",38.333333333333,38.333333333333,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-905","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70,70,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-905--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70,70,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1188","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.3333,70.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1399","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,33.6667,33.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1386","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-985","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (Reasoning)",null,"Claude Opus 4.5 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1032","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1032--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1019","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.3333,70.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1019--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.3333,70.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1171","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,67.6667,67.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1171--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,67.6667,67.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1769","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,60.6667,60.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1717","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,64.6667,64.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1371","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65.6667,65.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1009","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1009--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-957","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-957--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1625","command-a-plus","Command A+","Command A+",null,"Command A+","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,46,46,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1531","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65,65,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1516","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65,65,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1155","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63,63,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1155--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63,63,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1098","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1098--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1831","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,64.3333,64.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1791","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1129","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1207","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)",null,"Gemini 3 Pro Preview (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","excluded","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1207--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)","gemini-3-pro-high","Gemini 3 Pro Preview (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,70.6667,70.6667,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","excluded","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1070","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65.3333,65.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1213","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,72.6667,72.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-914","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.3333,69.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-914--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.3333,69.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-2043","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis Long Context Reasoning independent evaluation.",null,"Artificial Analysis Long Context Reasoning independent evaluation.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,62,62,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-LCR 62.0%."],["evidence-2026-07-2003","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis Long Context Reasoning independent evaluation.",null,"Artificial Analysis Long Context Reasoning independent evaluation.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.7,69.7,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-LCR 69.7%."],["evidence-2026-07-1875","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,55.3333,55.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1611","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,55.6667,55.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1224","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,62,62,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1732","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,54.3333,54.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1545","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,64,64,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1026","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63.3333,63.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-963","glm-5-turbo","GLM-5-Turbo","GLM-5-Turbo",null,"GLM-5-Turbo","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,60.6667,60.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1085","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,62.3333,62.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1250","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,71.3333,71.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1250--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,71.3333,71.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-980","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,61,61,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1331","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75.6,75.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1331--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75.6,75.6,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1318","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69,69,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1318--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69,69,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1345","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1345--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1063","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75,75,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1063--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75,75,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1297","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,72.6667,72.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1297--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,72.6667,72.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1308","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75.6667,75.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1308--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,75.6667,75.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1079","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)",null,"GPT-5.3 Codex (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1079--configuration--gpt-5-3-codex-xhigh","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)","gpt-5-3-codex-xhigh","GPT-5.3 Codex (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1046","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1046--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1281","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.3333,69.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1281--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.3333,69.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1144","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1144--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-969","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74.3333,74.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-969--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74.3333,74.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-935","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-935--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-942","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,73.6667,73.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-942--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,73.6667,73.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-991","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-991--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1864","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,50.3333,50.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1652","grok-4","Grok 4","Grok 4",null,"Grok 4","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,68,68,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1756","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,64.6667,64.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1441","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,58,58,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-1109","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,64.3333,64.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1137","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,67.6667,67.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1886","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,48.3333,48.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1843","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,52.3333,52.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1664","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-1180","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65.3333,65.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1409","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.6667,69.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1427","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.3333,66.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1947","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis Long Context Reasoning.",null,"Kimi K3; Artificial Analysis Long Context Reasoning.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74.7,74.7,"percent","higher","1.4.1","reference-only","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis AA-LCR score for Kimi K3."],["evidence-2026-07-1781","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,34.6667,34.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1581","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63,63,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1165","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.6667,66.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1092","mimo-v2-pro","MiMo-V2-Pro","MiMo-V2-Pro",null,"MiMo-V2-Pro","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,60.6667,60.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1568","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,62.6667,62.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-925","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,73.3333,73.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1745","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,61,61,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1684","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,59,59,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1558","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-1054","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,68.6667,68.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1000","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,74,74,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1039","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,34.6667,34.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1240","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,61,61,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-950","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,44.6667,44.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1261","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.6667,69.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1234","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63.3333,63.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1234--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63.3333,63.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1918","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63.3,63.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1918--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,63.3,63.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1818","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,60,60,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1596","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,67,67,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1638","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,54.3333,54.3333,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-aa-lcr-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,52,52,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1854","o1","o1","o1",null,"o1","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,59.3333,59.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1358","o3","o3","o3",null,"o3","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.3333,69.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1805","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,55,55,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1805--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,55,55,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1675","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66,66,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1492","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,66.6667,66.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1504","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,67.3333,67.3333,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1705","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,62.6667,62.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1474","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65.6667,65.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1695","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,52.6667,52.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1452","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,68.6667,68.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1464","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.6667,69.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-1268","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69.6667,69.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1120","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,69,69,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1198","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","aa-lcr","AA Long Context Reasoning","reasoning","Artificial Analysis",null,65,65,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-lcr-leaderboard","aa-lcr-leaderboard","AA Long Context Reasoning Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-LCR accuracy independently evaluated by Artificial Analysis."],["benchlm-ref-claude-opus-4-5-aineedle-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",74,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aineedle-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",74,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aineedle-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",74,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aineedle-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",63.3,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aineedle-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",63.3,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aineedle-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",63.3,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aineedle-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.7,50.4673,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aineedle-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.7,50.4673,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aineedle-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.7,50.4673,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aineedle-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.3,46.729,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aineedle-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.3,46.729,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aineedle-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","benchlm-aineedle","AI-Needle","reasoning","Qwen","2026",68.3,46.729,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi1-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arcagi1","ARC-AGI-1 Semi-Private Evaluation","reasoning","ARC Prize Foundation","2026",97.5,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi1-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","benchlm-arcagi1","ARC-AGI-1 Semi-Private Evaluation","reasoning","ARC Prize Foundation","2026",97.5,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-573","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",45.1,45.1,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-575","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",31.1,31.1,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-018","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",77.1,77.1,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-230","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",77.1,77.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-009","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",72.1,72.1,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-231","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",72.1,72.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-572","gpt-5-2","GPT-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpt-5-2-pro-unspecified","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",54.2,54.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-028","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",84.6,84.6,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-229","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",85,85,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-574","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2",42.5,42.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-claude-opus-4-7-adaptive-arcagi2-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",75.8,87.1148,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-arcagi2-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",75.8,78.834,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-arcagi2-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",75.8,78.834,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-arcagi2-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",72.08,74.1191,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-arcagi2-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",72.08,74.1191,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi2-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",90.4,97.3384,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi2-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",90.4,97.3384,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-arcagi2-2026-07-21","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",13.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-arcagi2-2026-07-27","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",13.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-5-arcagi2-2026-08-01","claude-sonnet-4-5","Claude Sonnet 4.5","Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",13.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-07-21","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",45.1,44.1176,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-07-27","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",45.1,39.924,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-08-01","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",45.1,39.924,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-arcagi2-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",31.1,24.5098,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-arcagi2-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",31.1,22.18,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-arcagi2-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",31.1,22.18,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-arcagi2-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",77.1,88.9356,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-arcagi2-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",77.08,80.4563,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-arcagi2-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",77.08,80.4563,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-arcagi2-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",72.1,81.9328,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-arcagi2-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",72.1,74.1445,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-arcagi2-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",72.1,74.1445,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-arcagi2-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",52.9,55.042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-arcagi2-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",52.9,49.8099,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-arcagi2-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",52.9,49.8099,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-arcagi2-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",83.3,97.619,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-arcagi2-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",83.3,88.3397,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-arcagi2-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",83.3,88.3397,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-arcagi2-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",73.95,76.4892,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-arcagi2-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",73.95,76.4892,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-arcagi2-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",85,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-arcagi2-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",85,90.4943,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-arcagi2-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",85,90.4943,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-arcagi2-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",59.54,58.2256,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-arcagi2-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",59.54,58.2256,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-arcagi2-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",92.5,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-arcagi2-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",92.5,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-arcagi2-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",83.9,89.1001,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-arcagi2-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",83.9,89.1001,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-arcagi2-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",53.3,55.6022,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-arcagi2-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",53.3,50.3169,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-arcagi2-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",53.3,50.3169,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-arcagi2-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",52.64,49.4804,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-arcagi2-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",52.64,49.4804,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-arcagi2-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",40.1,33.5868,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-arcagi2-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",42.5,40.4762,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-arcagi2-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",42.5,36.6286,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-arcagi2-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","arc-agi-2","ARC-AGI-2","reasoning","ARC Prize Foundation","2025",42.5,36.6286,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-arcagi3-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.18,0.2993,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-arcagi3-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.18,0.2993,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-arcagi3-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",1.52,4.7556,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-arcagi3-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",1.52,4.7556,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi3-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",30.16,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-arcagi3-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",30.16,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-arcagi3-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.42,1.0974,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-arcagi3-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.42,1.0974,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-arcagi3-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.21,0.3991,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-arcagi3-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.21,0.3991,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-arcagi3-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.43,1.1307,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-arcagi3-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.43,1.1307,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-arcagi3-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-arcagi3-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.18,0.2993,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-arcagi3-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.18,0.2993,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-arcagi3-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",7.8,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-arcagi3-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",7.78,25.5737,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-arcagi3-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",7.78,25.5737,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-arcagi3-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.8,7.8947,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-arcagi3-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.8,2.3612,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-arcagi3-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.8,2.3612,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-arcagi3-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.09,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-arcagi3-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.09,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-arcagi3-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.3,0.6984,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-arcagi3-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","2026",0.3,0.6984,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-2056","claude-opus-5","Claude Opus 5","ARC-AGI-3 novel problem-solving score as published in the Anthropic Claude Opus 5 launch table (high effort).",null,"ARC-AGI-3 novel problem-solving score as published in the Anthropic Claude Opus 5 launch table (high effort).","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","3",30.2,30.2,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 30.2% on ARC-AGI-3; launch copy states roughly 3× the next-best model in their comparison set."],["evidence-2026-07-2056--configuration--claude-opus-5-high","claude-opus-5","Claude Opus 5","ARC-AGI-3 novel problem-solving score as published in the Anthropic Claude Opus 5 launch table (high effort).","claude-opus-5-high","ARC-AGI-3 novel problem-solving score as published in the Anthropic Claude Opus 5 launch table (high effort).","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","3",30.2,30.2,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 30.2% on ARC-AGI-3; launch copy states roughly 3× the next-best model in their comparison set."],["evidence-2026-07-305","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","3",0.2,0.2,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-303","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","3",7.8,7.8,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-304","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","arc-agi-3","ARC-AGI-3","reasoning","ARC Prize Foundation","3",0.8,0.8,"percent","higher","1.2.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant."],["evidence-2026-07-522","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",67.5,67.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-455","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",87.8,87.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-510","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",80.9,80.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-549","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",62.2,62.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-539","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",88.3,88.3,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-468","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",81.8,81.8,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-422","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",72.7,72.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-544","gemma-4-31b","Gemma 4 31B","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",52.3,52.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-490","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",56.2,56.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-461","glm-5-1","GLM-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",56.2,56.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-516","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",65.4,65.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-485","gpt-5-3-codex","GPT-5.3-Codex","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",87.4,87.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-428","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",87.5,87.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-479","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",29.3,29.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-416","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",81.9,81.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-557","grok-4-1","Grok 4.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",90.8,90.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-442","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",50.4,50.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-504","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-reasoning","BenchLM Reasoning prior","reasoning","BenchLM","bench-align-v5.1",43.6,43.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (reasoning) used to estimate missing category coverage under methodology 1.3.0."],["benchlm-ref-deepseek-v4-flash-base-bbh-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",86.9,98.2609,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-bbh-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",86.9,98.2609,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-bbh-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",86.9,98.2609,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bbh-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",87.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bbh-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",87.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-bbh-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",87.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-bbh-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",53,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-bbh-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",53,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-bbh-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",53,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bbh-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",71.89,54.7536,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bbh-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",71.89,54.7536,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-bbh-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",71.89,54.7536,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-bbh-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",78.8,74.7826,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-bbh-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",78.8,74.7826,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-bbh-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-bbh","BIG-Bench Hard","reasoning","Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","2022",78.8,74.7826,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",82.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",82.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",82.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",85.2,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",85.2,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-cluewsc","CLUEWSC","reasoning","DeepSeek-AI","2026",85.2,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-corpusqa1m-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",59.3,94.1935,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",15.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",15.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",15.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",60.5,96.7742,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-corpusqa1m-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",56.5,88.172,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",35.6,43.2258,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",35.6,43.2258,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",35.6,43.2258,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-corpusqa1m","CorpusQA 1M","reasoning","DeepSeek-AI","2026",62,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-critpt-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-critpt-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-critpt-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-critpt-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-critpt-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-critpt-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-critpt-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-critpt-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-critpt-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-critpt-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",28.6,88.5449,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-critpt-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",28.6,88.5449,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-critpt-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",28.6,88.5449,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-critpt-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-critpt-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-critpt-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-critpt-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-critpt-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-critpt-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-critpt-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.6,39.0093,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-critpt-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.6,39.0093,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-critpt-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.6,39.0093,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-critpt-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.8,8.6687,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-critpt-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.8,8.6687,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-critpt-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.8,8.6687,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-critpt-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12,37.1517,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-critpt-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12,37.1517,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-critpt-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12,37.1517,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-critpt-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.1,15.7895,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-critpt-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.1,15.7895,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-critpt-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.1,15.7895,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-critpt-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-critpt-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-critpt-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-critpt-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",29.1,90.0929,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-critpt-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",29.1,90.0929,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-critpt-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-critpt-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-critpt-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-critpt-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-critpt-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-critpt-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-critpt-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-critpt-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-critpt-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-critpt-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-critpt-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-critpt-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-critpt-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-critpt-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-critpt-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-critpt-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-critpt-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-critpt-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-critpt-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-critpt-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-critpt-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-critpt-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.4,10.5263,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-critpt-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",7.1,21.9814,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-critpt-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-critpt-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",12.9,39.9381,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-critpt-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-critpt-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-critpt-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-critpt-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-critpt-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-critpt-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-critpt-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-critpt-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-critpt-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-critpt-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-critpt-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-critpt-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-critpt-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.6,8.0495,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-critpt-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.6,8.0495,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-critpt-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.6,8.0495,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-critpt-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-critpt-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-critpt-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-critpt-2026-07-21","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",25.7,79.5666,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-critpt-2026-07-27","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",25.7,79.5666,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-deep-think-critpt-2026-08-01","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",25.7,79.5666,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-critpt-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-critpt-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-critpt-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-critpt-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-critpt-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-critpt-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-critpt-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",17.7,54.7988,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-critpt-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",17.7,54.7988,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-critpt-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",17.7,54.7988,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-critpt-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.1,40.5573,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-critpt-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.1,40.5573,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-critpt-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.1,40.5573,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-critpt-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-critpt-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-critpt-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-critpt-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10.6,32.8173,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-critpt-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10.6,32.8173,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-critpt-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10.6,32.8173,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-critpt-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-critpt-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-critpt-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-critpt-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-critpt-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-critpt-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-critpt-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-critpt-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-critpt-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-critpt-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-critpt-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-critpt-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-critpt-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-critpt-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-critpt-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-critpt-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-critpt-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-critpt-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-critpt-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-critpt-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-critpt-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-critpt-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-critpt-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-critpt-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-critpt-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-critpt-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-critpt-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-critpt-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-critpt-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-critpt-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-critpt-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-critpt-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-critpt-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-critpt-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-critpt-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-critpt-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.6,14.2415,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-critpt-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-critpt-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-critpt-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.9,64.7059,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-critpt-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-critpt-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-critpt-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-critpt-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-critpt-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-critpt-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-critpt-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-critpt-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-critpt-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-critpt-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-critpt-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-critpt-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-critpt-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-critpt-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-critpt-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-critpt-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-critpt-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-critpt-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-critpt-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-critpt-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-critpt-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-critpt-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-critpt-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-critpt-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-critpt-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-critpt-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.7,17.6471,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-critpt-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.6,35.9133,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-critpt-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.6,35.9133,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-critpt-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.6,35.9133,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-critpt-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8.7,26.935,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-critpt-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8.7,26.935,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-critpt-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8.7,26.935,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-critpt-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-critpt-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-critpt-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",16.9,52.322,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-critpt-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-critpt-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-critpt-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-critpt-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-critpt-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-critpt-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-critpt-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-critpt-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-critpt-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-critpt-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.3,28.7926,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-critpt-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.3,28.7926,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-critpt-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.3,28.7926,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-critpt-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30.6,94.7368,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-critpt-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30.6,94.7368,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-critpt-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30.6,94.7368,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-critpt-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",27.1,83.9009,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-critpt-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",27.1,83.9009,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-critpt-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",27.1,83.9009,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-critpt-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.6,63.7771,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-critpt-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.6,63.7771,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-critpt-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",20.6,63.7771,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-critpt-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",32.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-critpt-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",32.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-critpt-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",32.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-critpt-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-critpt-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-critpt-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",30,92.8793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-critpt-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-critpt-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-critpt-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-critpt-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-critpt-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-critpt-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.4,4.3344,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-critpt-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-critpt-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-critpt-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-critpt-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-critpt-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-critpt-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-critpt-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-critpt-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-critpt-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-critpt-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-critpt-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-critpt-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-critpt-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-critpt-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-critpt-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2,6.192,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-critpt-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-critpt-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-critpt-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-critpt-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-critpt-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-critpt-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-critpt-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-critpt-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-critpt-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-critpt-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.4,47.678,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-critpt-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.4,47.678,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-critpt-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.4,47.678,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-critpt-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-critpt-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-critpt-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-critpt-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-critpt-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-critpt-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-critpt-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-critpt-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-critpt-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4.9,15.1703,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-critpt-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.4,16.7183,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-critpt-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.4,16.7183,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-critpt-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",5.4,16.7183,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-critpt-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8.3,25.6966,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-critpt-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-critpt-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-critpt-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-critpt-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-critpt-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-critpt-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-critpt-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-critpt-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-critpt-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-critpt-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-critpt-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-critpt-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-critpt-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-critpt-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-critpt-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",8,24.7678,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-critpt-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-critpt-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-critpt-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",10,30.9598,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-critpt-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-critpt-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-critpt-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",23.4,72.4458,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-critpt-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-critpt-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-critpt-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-critpt-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-critpt-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-critpt-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-critpt-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-critpt-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-critpt-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-critpt-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-critpt-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-critpt-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-critpt-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-critpt-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-critpt-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-critpt-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-critpt-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-critpt-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-critpt-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-critpt-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-critpt-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-critpt-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-critpt-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-critpt-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-critpt-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4,12.3839,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-critpt-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4,12.3839,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-critpt-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",4,12.3839,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-critpt-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-critpt-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-critpt-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-critpt-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-critpt-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-critpt-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-critpt-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-critpt-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-critpt-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-critpt-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-critpt-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-critpt-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-critpt-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-critpt-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-critpt-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-critpt-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-critpt-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-critpt-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-critpt-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-critpt-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-critpt-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-critpt-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-critpt-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-critpt-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-critpt-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.3,34.9845,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-critpt-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.3,34.9845,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-critpt-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",11.3,34.9845,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-critpt-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.1,46.7492,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-critpt-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.1,46.7492,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-critpt-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",15.1,46.7492,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-critpt-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-critpt-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-critpt-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-critpt-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-critpt-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-critpt-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.1,9.5975,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-critpt-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-critpt-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-critpt-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-critpt-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-critpt-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-critpt-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-critpt-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-critpt-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-critpt-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-critpt-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-critpt-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-critpt-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-critpt-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-critpt-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-critpt-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-critpt-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-critpt-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-critpt-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",3.7,11.4551,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-critpt-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-critpt-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-critpt-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-critpt-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-critpt-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-critpt-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-critpt-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-critpt-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-critpt-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-critpt-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-critpt-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-critpt-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.7,5.2632,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-critpt-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-critpt-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-critpt-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.6,1.8576,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-critpt-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-critpt-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-critpt-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-critpt-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-critpt-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-critpt-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",1.1,3.4056,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-critpt-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-critpt-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-critpt-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.9,8.9783,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-critpt-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-critpt-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-critpt-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-critpt-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.4,41.4861,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-critpt-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.4,41.4861,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-critpt-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",13.4,41.4861,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-critpt-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-critpt-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-critpt-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",9.1,28.1734,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-critpt-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-critpt-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-critpt-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-critpt-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-critpt-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-critpt-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.3,0.9288,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-critpt-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-critpt-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-critpt-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-critpt-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.3,7.1207,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-critpt-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.3,7.1207,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-critpt-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",2.3,7.1207,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-critpt-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-critpt-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-critpt-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-critpt-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-critpt-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-critpt-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","critpt","CritPt","reasoning","CritPt authors","2026",0.9,2.7864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["aa-individual:claude-fable-5:critpt:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",28.571428570000002,28.571428570000002,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:claude-opus-4-6-thinking:critpt:2026-08-29","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",12.571428571429,12.571428571429,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-6-thinking-2026-08-29","aa-current-claude-opus-4-6-thinking-2026-08-29","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-6-adaptive","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:claude-opus-4-7-adaptive:critpt:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",12,12,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:claude-opus-4-8:critpt:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",20.8571428571429,20.8571428571429,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:critpt:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",29.1428571428571,29.1428571428571,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-sonnet-5:critpt:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",16.8571428571429,16.8571428571429,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:deepseek-v3-1:critpt:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","deepseek-v3-1-non-reasoning","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-2026-08-29","aa-current-deepseek-v3-1-2026-08-29","DeepSeek V3.1 (Non-reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:deepseek-v3-1-reasoning:critpt:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","deepseek-v3-1-reasoning-default","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",2,2,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-reasoning-2026-08-29","aa-current-deepseek-v3-1-reasoning-2026-08-29","DeepSeek V3.1 (Reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:deepseek-v4-flash-vision-exp:critpt:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",10.8571428571429,10.8571428571429,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual critpt result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-individual:deepseek-v4-pro-0813:critpt:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",18,18,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-6-flash:critpt:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",10.5714285714286,10.5714285714286,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:critpt:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",9.42857142857143,9.42857142857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:critpt:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gemma-4-26b-a4b:critpt:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:glm-5-2:critpt:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",20.857142857143,20.857142857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:critpt:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",19.1428571428571,19.1428571428571,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:critpt:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",15.4285714285714,15.4285714285714,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual critpt result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:critpt:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:critpt:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:gpt-5-4:critpt:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",23.42857143,23.42857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gpt-5-4-pro:critpt:2026-08-29","gpt-5-4","GPT-5.4","GPT-5.4 Pro (xhigh) (Artificial Analysis completed independent run)","gpt-5-4-pro-xhigh","GPT-5.4 Pro (xhigh) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",30,30,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-gpt-5-4-pro-2026-08-29","aa-current-gpt-5-4-pro-2026-08-29","GPT-5.4 Pro (xhigh) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4-pro","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:gpt-5-5:critpt:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",27.142857142857103,27.142857142857103,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-luna:critpt:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",20.571428571428598,20.571428571428598,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-sol:critpt:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",32.2857142857143,32.2857142857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-terra:critpt:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",30,30,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-5:critpt:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",15.4285714285714,15.4285714285714,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-6:critpt:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",17.1428571428571,17.1428571428571,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:kimi-k2-5-reasoning:critpt:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",3.142857142857,3.142857142857,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:kimi-k3:critpt:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",23.4285714285714,23.4285714285714,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:llama-4-maverick:critpt:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:critpt:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:critpt:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",2.571428571429,2.571428571429,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-current:mistral-medium-3-5-128b:critpt:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0,0,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:mistral-small-4-reasoning:critpt:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0.285714285714,0.285714285714,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:muse-spark-1-1:critpt:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",15.1428571428571,15.1428571428571,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:muse-spark-1-2:critpt:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",17.7142857142857,17.7142857142857,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual critpt result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:critpt:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0.857142857143,0.857142857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-397b-reasoning:critpt:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",1.70612244898,1.70612244898,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:critpt:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0.571428571429,0.571428571429,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-6-35b-a3b:critpt:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0.285714285714,0.285714285714,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen-3-8-flash-next:critpt:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","critpt","CritPt","reasoning","CritPt authors","standard",11.142857142857,11.142857142857,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:critpt:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","critpt","CritPt","reasoning","CritPt authors","standard",0.857142857143,0.857142857143,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-07-907","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","critpt","CritPt","reasoning","CritPt authors",null,28.5714,28.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-907--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","critpt","CritPt","reasoning","CritPt authors",null,28.5714,28.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1190","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1387","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-987","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (Reasoning)",null,"Claude Opus 4.5 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,4.5714,4.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1034","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12.5714,12.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1034--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12.5714,12.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1021","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12,12,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1021--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12,12,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1173","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,20.8571,20.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1173--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,20.8571,20.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1771","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.8571,0.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1719","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1373","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1011","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,3.1429,3.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1011--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,3.1429,3.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-959","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,16.8571,16.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-959--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,16.8571,16.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1627","command-a-plus","Command A+","Command A+",null,"Command A+","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1533","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.7143,1.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1518","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,2.8571,2.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1157","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,7.1429,7.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1157--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,7.1429,7.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1100","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12.8571,12.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1100--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","critpt","CritPt","reasoning","CritPt authors",null,12.8571,12.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1833","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1793","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","critpt","CritPt","reasoning","CritPt authors",null,2.5714,2.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1131","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,8.5714,8.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1209","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)",null,"Gemini 3 Pro Preview (high)","critpt","CritPt","reasoning","CritPt authors",null,9.1429,9.1429,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","excluded","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1209--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)","gemini-3-pro-high","Gemini 3 Pro Preview (high)","critpt","CritPt","reasoning","CritPt authors",null,9.1429,9.1429,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","excluded","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1072","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1215","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","critpt","CritPt","reasoning","CritPt authors",null,17.7143,17.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-916","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","critpt","CritPt","reasoning","CritPt authors",null,13.1429,13.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-916--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","critpt","CritPt","reasoning","CritPt authors",null,13.1429,13.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-2044","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis CritPt independent evaluation.",null,"Artificial Analysis CritPt independent evaluation.","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","CritPt 0.0% as listed on AA for 3.5 Flash-Lite."],["evidence-2026-07-2004","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis CritPt independent evaluation.",null,"Artificial Analysis CritPt independent evaluation.","critpt","CritPt","reasoning","CritPt authors",null,10.6,10.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","CritPt 10.6%."],["evidence-2026-07-1877","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1613","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1226","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.4286,1.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1734","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1547","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.7102,1.7102,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1028","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,2,2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-965","glm-5-turbo","GLM-5-Turbo","GLM-5-Turbo",null,"GLM-5-Turbo","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1087","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,4.5714,4.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1252","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","critpt","CritPt","reasoning","CritPt authors",null,20.8571,20.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1252--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","critpt","CritPt","reasoning","CritPt authors",null,20.8571,20.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-982","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1333","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","critpt","CritPt","reasoning","CritPt authors",null,5.7143,5.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1333--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","critpt","CritPt","reasoning","CritPt authors",null,5.7143,5.7143,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1320","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","critpt","CritPt","reasoning","CritPt authors",null,5.1429,5.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1320--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","critpt","CritPt","reasoning","CritPt authors",null,5.1429,5.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1347","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","critpt","CritPt","reasoning","CritPt authors",null,1.4286,1.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1347--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","critpt","CritPt","reasoning","CritPt authors",null,1.4286,1.4286,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1065","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","critpt","CritPt","reasoning","CritPt authors",null,4.8571,4.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1065--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","critpt","CritPt","reasoning","CritPt authors",null,4.8571,4.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1299","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,11.551,11.551,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1299--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,11.551,11.551,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1310","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,8.6694,8.6694,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1310--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,8.6694,8.6694,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1081","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)",null,"GPT-5.3 Codex (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,16.8571,16.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1081--configuration--gpt-5-3-codex-xhigh","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)","gpt-5-3-codex-xhigh","GPT-5.3 Codex (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,16.8571,16.8571,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1048","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,23.4286,23.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1048--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,23.4286,23.4286,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1283","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,10,10,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1283--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,10,10,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1146","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,9.2531,9.2531,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1146--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,9.2531,9.2531,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-971","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,27.1429,27.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-971--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,27.1429,27.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1248","gpt-5-5","GPT-5.5","GPT-5.5 Pro (xhigh)","gpt-5-5-pro-default-high","GPT-5.5 Pro (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,30.5714,30.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1248--configuration--gpt-5-5-pro-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 Pro (xhigh)","gpt-5-5-pro","GPT-5.5 Pro (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,30.5714,30.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-937","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","critpt","CritPt","reasoning","CritPt authors",null,20.5714,20.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-937--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","critpt","CritPt","reasoning","CritPt authors",null,20.5714,20.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-944","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","critpt","CritPt","reasoning","CritPt authors",null,32.2857,32.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-944--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","critpt","CritPt","reasoning","CritPt authors",null,32.2857,32.2857,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-993","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","critpt","CritPt","reasoning","CritPt authors",null,30,30,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-993--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","critpt","CritPt","reasoning","CritPt authors",null,30,30,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1866","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1654","grok-4","Grok 4","Grok 4",null,"Grok 4","critpt","CritPt","reasoning","CritPt authors",null,2,2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1758","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,2.8571,2.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1443","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,6.5714,6.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-1111","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","critpt","CritPt","reasoning","CritPt authors",null,8,8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1139","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","critpt","CritPt","reasoning","CritPt authors",null,15.4286,15.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1888","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1845","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1666","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","critpt","CritPt","reasoning","CritPt authors",null,2.5714,2.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-1182","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,3.1429,3.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1411","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","critpt","CritPt","reasoning","CritPt authors",null,8,8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1429","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","critpt","CritPt","reasoning","CritPt authors",null,10,10,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1948","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis CritPt evaluation.",null,"Kimi K3; Artificial Analysis CritPt evaluation.","critpt","CritPt","reasoning","CritPt authors",null,23.4,23.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","Artificial Analysis CritPt score for Kimi K3."],["evidence-2026-07-1783","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1583","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,4.2857,4.2857,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1167","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1094","mimo-v2-pro","MiMo-V2-Pro","MiMo-V2-Pro",null,"MiMo-V2-Pro","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1570","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","critpt","CritPt","reasoning","CritPt authors",null,3.7143,3.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-927","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","critpt","CritPt","reasoning","CritPt authors",null,4,4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1747","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","critpt","CritPt","reasoning","CritPt authors",null,0.8571,0.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1686","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1560","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-1056","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1002","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","critpt","CritPt","reasoning","CritPt authors",null,3.7143,3.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1041","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1242","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-952","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1263","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","critpt","CritPt","reasoning","CritPt authors",null,11.3306,11.3306,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1236","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,15.1429,15.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1236--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","critpt","CritPt","reasoning","CritPt authors",null,15.1429,15.1429,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1919","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","critpt","CritPt","reasoning","CritPt authors",null,15.1,15.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-16","2026-07-15","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1919--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","critpt","CritPt","reasoning","CritPt authors",null,15.1,15.1,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-16","2026-07-15","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1820","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,3.1429,3.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1598","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,3.1429,3.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1640","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","critpt","CritPt","reasoning","CritPt authors",null,0,0,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1856","o1","o1","o1",null,"o1","critpt","CritPt","reasoning","CritPt authors",null,0.2857,0.2857,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1360","o3","o3","o3",null,"o3","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1807","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1807--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1677","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","critpt","CritPt","reasoning","CritPt authors",null,1.7143,1.7143,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1494","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.5714,0.5714,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1506","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.8531,0.8531,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1707","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,0.8571,0.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1476","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.7061,1.7061,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1697","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","critpt","CritPt","reasoning","CritPt authors",null,0.5633,0.5633,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1454","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","critpt","CritPt","reasoning","CritPt authors",null,1.1429,1.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1466","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","critpt","CritPt","reasoning","CritPt authors",null,3.7061,3.7061,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-1270","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","critpt","CritPt","reasoning","CritPt authors",null,2.8571,2.8571,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1122","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","critpt","CritPt","reasoning","CritPt authors",null,13.4286,13.4286,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["evidence-2026-07-1200","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","critpt","CritPt","reasoning","CritPt authors",null,9.1429,9.1429,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-critpt-leaderboard","aa-critpt-leaderboard","CritPt Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/critpt","2026-07-15","2026-07-15","2026-07-15","source-checked","CritPt accuracy independently evaluated by Artificial Analysis."],["benchlm-ref-deepseek-v4-flash-base-drop-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.6,99.5495,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-drop-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.6,99.5495,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-drop-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.6,99.5495,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-drop-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.7,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-drop-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-drop-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",88.7,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-drop-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",66.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-drop-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",66.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-drop-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","benchlm-drop","Discrete Reasoning Over Paragraphs","reasoning","DeepSeek-AI","2026",66.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-genebenchpro-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-genebenchpro","GeneBench-Pro","reasoning","OpenAI","2026",28.7,50,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-genebenchpro-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-genebenchpro","GeneBench-Pro","reasoning","OpenAI","2026",28.7,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-genebenchpro-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","benchlm-genebenchpro","GeneBench-Pro","reasoning","OpenAI","2026",28.7,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["epoch-gpqa-mTa4f52zsEuxdrPy7WfQvX","claude-2","Claude 2.0","claude-2.0","claude-2-epoch-claude-2-0","claude-2.0","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",34.659091,34.659091,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.35±0.03; stderr=0.025076255983551475."],["epoch-gpqa-eMAisrJd7fAUHPfcPWXeXo","claude-21","Claude 2.1","claude-2.1","claude-21-epoch-claude-2-1","claude-2.1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",32.954545,32.954545,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.33±0.02; stderr=0.02324885083265289."],["epoch-gpqa-T3FWdCCWBryVjxwTcoY5bK","claude-3-haiku","Claude 3 Haiku","Claude 3 Haiku","claude-3-haiku-epoch-claude-3-haiku-20240307","Claude 3 Haiku","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",36.300505,36.300505,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.36±0.02; stderr=0.024180910672074975."],["epoch-gpqa-hTnDa4gn8RYduCEYqoNywe","claude-3-opus","Claude 3 Opus","Claude 3 Opus","claude-3-opus-epoch-claude-3-opus-20240229","Claude 3 Opus","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",47.159091,47.159091,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.47±0.03; stderr=0.026395393956533515."],["epoch-gpqa-QLejPZXhjX6SQCQgzjNvCG","claude-3-sonnet","Claude 3 Sonnet","Claude 3 Sonnet","claude-3-sonnet-epoch-claude-3-sonnet-20240229","Claude 3 Sonnet","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",40.593434,40.593434,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.41±0.02; stderr=0.023920223491748827."],["epoch-gpqa-JpLxrpvthx7A75sbaHZ9Et","claude-3-5-haiku","Claude 3.5 Haiku","Claude 3.5 Haiku (Oct 2024)","claude-3-5-haiku-epoch-claude-3-5-haiku-20241022","Claude 3.5 Haiku (Oct 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",38.131313,38.131313,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-03-12","2025-03-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-03-12","2026-09-01","2026-08-18","source-checked","choice:0.38±0.03; stderr=0.026351515390511556."],["epoch-gpqa-78Wpvp8WDByAcQRBas72cs","claude-3-5-sonnet","Claude 3.5 Sonnet","Claude 3.5 Sonnet (Jun 2024)","claude-3-5-sonnet-epoch-claude-3-5-sonnet-20240620","Claude 3.5 Sonnet (Jun 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",54.040404,54.040404,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.54±0.03; stderr=0.027528555453348653."],["epoch-gpqa-GEaAdb9UmoUFbEtBx7WzAR","claude-haiku-4-5","Claude Haiku 4.5","Claude Haiku 4.5 (32k thinking)","claude-haiku-4-5-epoch-claude-haiku-4-5-20251001-32k","Claude Haiku 4.5 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",71.212121,71.212121,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-22","2025-10-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-22","2026-09-01","2026-08-17","source-checked","choice:0.71±0.03; stderr=0.03225883512300997."],["epoch-gpqa-cgnopvpB6mzsfb9yGnYK2z","claude-haiku-4-5","Claude Haiku 4.5","claude-haiku-4-5-20251001","claude-haiku-4-5-epoch-claude-haiku-4-5-20251001","claude-haiku-4-5-20251001","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",60.479798,60.479798,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-16","2025-10-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-16","2026-09-01","2026-08-17","source-checked","choice:0.6±0.03; stderr=0.028154535023020826."],["epoch-gpqa-EnjJ8bGi9vRNNQFzfkLCkQ","claude-opus-4","Claude Opus 4","Claude Opus 4","claude-opus-4-epoch-claude-opus-4-20250514","Claude Opus 4","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",69.191919,69.191919,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-22","2025-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-22","2026-09-01","2026-08-18","source-checked","choice:0.69±0.03; stderr=0.03289477330098615."],["epoch-gpqa-CtxFusPxCraL3EeRJu2c5k","claude-opus-4","Claude Opus 4","Claude Opus 4 (16k thinking)","claude-opus-4-epoch-claude-opus-4-20250514-16k","Claude Opus 4 (16k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",76.262626,76.262626,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-22","2025-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-22","2026-09-01","2026-08-18","source-checked","choice:0.76±0.03; stderr=0.03031371053819892."],["epoch-gpqa-j62bkFDjBSeFyVpnGzuzg4","claude-opus-4-1","Claude Opus 4.1","Claude Opus 4.1","claude-opus-4-1-epoch-claude-opus-4-1-20250805","Claude Opus 4.1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",73.232323,73.232323,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-05","2025-08-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-05","2026-09-01","2026-08-18","source-checked","choice:0.73±0.03; stderr=0.031544498882702825."],["epoch-gpqa-erLgyKQs2mg2memE9SJvK5","claude-opus-4-1","Claude Opus 4.1","Claude Opus 4.1 (16k thinking)","claude-opus-4-1-epoch-claude-opus-4-1-20250805-16k","Claude Opus 4.1 (16k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",77.272727,77.272727,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-05","2025-08-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-05","2026-09-01","2026-08-18","source-checked","choice:0.77±0.03; stderr=0.029857515673386438."],["epoch-gpqa-3obcsGCfuwYeEm9wRyqzPg","claude-opus-4-1","Claude Opus 4.1","Claude Opus 4.1 (27k thinking)","claude-opus-4-1-epoch-claude-opus-4-1-20250805-27k","Claude Opus 4.1 (27k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",76.767677,76.767677,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-05","2025-08-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-05","2026-09-01","2026-08-18","source-checked","choice:0.77±0.03; stderr=0.030088629490217445."],["epoch-gpqa-ZF4eejw7J2KrcoNWd6hUbU","claude-sonnet-4","Claude Sonnet 4","Claude Sonnet 4","claude-sonnet-4-epoch-claude-sonnet-4-20250514","Claude Sonnet 4","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",66.666667,66.666667,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-22","2025-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-22","2026-09-01","2026-08-18","source-checked","choice:0.67±0.03; stderr=0.033586181457325226."],["epoch-gpqa-PPLoBnbKQ9CEYZNSk3E7LE","claude-sonnet-4","Claude Sonnet 4","Claude Sonnet 4 (16k thinking)","claude-sonnet-4-epoch-claude-sonnet-4-20250514-16k","Claude Sonnet 4 (16k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",75.757576,75.757576,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-22","2025-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-22","2026-09-01","2026-08-18","source-checked","choice:0.76±0.03; stderr=0.030532892233932022."],["epoch-gpqa-8hGbWtbpUjAQpebkKQfSi7","claude-sonnet-4","Claude Sonnet 4","Claude Sonnet 4 (32k thinking)","claude-sonnet-4-epoch-claude-sonnet-4-20250514-32k","Claude Sonnet 4 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",78.282828,78.282828,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-22","2025-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-22","2026-09-01","2026-08-18","source-checked","choice:0.78±0.03; stderr=0.02937661648494561."],["epoch-gpqa-J9wjcQqKJ7hCSYnnoa5E9i","claude-sonnet-4","Claude Sonnet 4","Claude Sonnet 4 (59k thinking)","claude-sonnet-4-epoch-claude-sonnet-4-20250514-59k","Claude Sonnet 4 (59k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",79.187817,79.187817,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-26","2025-05-26","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-26","2026-09-01","2026-08-18","source-checked","choice:0.79±0.03,choice:0.76±0.02; stderr=0.02674821434321096."],["epoch-gpqa-LkVuUxvsHNzpcg3sUiH7f3","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (16k thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929-16k","Claude Sonnet 4.5 (16k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",78.787879,78.787879,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-28","2025-10-28","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-28","2026-09-01","2026-08-17","source-checked","choice:0.79±0.03; stderr=0.02912652283458678."],["epoch-gpqa-ii6JE9j57eSbuswEsJArTa","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (32k thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929-32k","Claude Sonnet 4.5 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",81.725888,81.725888,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-21","2025-10-21","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-21","2026-09-01","2026-08-17","source-checked","choice:0.82±0.03; stderr=0.027603867020557334."],["epoch-gpqa-XaGuWDxPtqVvyxqdMncKFj","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (59k thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929-59k","Claude Sonnet 4.5 (59k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",82.323232,82.323232,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-28","2025-10-28","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-28","2026-09-01","2026-08-17","source-checked","choice:0.82±0.03; stderr=0.027178752639044908."],["epoch-gpqa-fsZmzw8jjyUgMQGwxodBwN","claude-sonnet-4-5","Claude Sonnet 4.5","Claude Sonnet 4.5 (no thinking)","claude-sonnet-4-5-epoch-claude-sonnet-4-5-20250929","Claude Sonnet 4.5 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",73.737374,73.737374,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-09-29","2025-09-29","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-09-29","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03; stderr=0.031353050095330834."],["epoch-gpqa-TTa3hu7PgRxHgizVreZsdJ","dbrx-instruct","DBRX Instruct","DBRX (instruct)","dbrx-instruct-epoch-dbrx-instruct","DBRX (instruct)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",32.891414,32.891414,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.33±0.03; stderr=0.03256944333564935."],["epoch-gpqa-f6pnAATjxmrFJS353EjQKi","deepseek-r1-distill-llama-70b","DeepSeek R1 Distill Llama 70B","DeepSeek-R1-Distill-Llama-70B","deepseek-r1-distill-llama-70b-epoch-deepseek-r1-distill-llama-70b","DeepSeek-R1-Distill-Llama-70B","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",55.744949,55.744949,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-03-10","2025-03-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-03-10","2026-09-01","2026-08-18","source-checked","choice:0.56±0.03; stderr=0.02973465219179067."],["epoch-gpqa-8UqaLobhmNJLuPGTTpeaCw","deepseek-r1-distill-qwen-14b","DeepSeek R1 Distill Qwen 14B","DeepSeek-R1-Distill-Qwen-14B","deepseek-r1-distill-qwen-14b-epoch-deepseek-r1-distill-qwen-14b","DeepSeek-R1-Distill-Qwen-14B","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",44.69697,44.69697,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-03-10","2025-03-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-03-10","2026-09-01","2026-08-18","source-checked","choice:0.45±0.03; stderr=0.02863094678190378."],["epoch-gpqa-Vwo3nMA8g2gGBCyTdKFMQJ","deepseek-v3","DeepSeek V3","DeepSeek-V3","deepseek-v3-epoch-deepseek-v3","DeepSeek-V3","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",56.534091,56.534091,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.57±0.03; stderr=0.02763990029950893."],["epoch-gpqa-fFGyQR2mfDufvrSV2rzWyV","deepseek-r1","DeepSeek-R1","DeepSeek-R1","deepseek-r1-epoch-deepseek-r1","DeepSeek-R1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",71.717172,71.717172,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-26","2025-05-26","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-26","2026-09-01","2026-08-18","source-checked","choice:0.72±0.03,choice:0.67±0.03; stderr=0.03070485693321271."],["epoch-gpqa-ibpS8bs4z3wD3Hb6F3kpdV","gemini-1-0-pro","Gemini 1.0 Pro","gemini-1.0-pro-001","gemini-1-0-pro-epoch-gemini-1-0-pro-001","gemini-1.0-pro-001","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",33.964646,33.964646,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.34±0.02; stderr=0.020110933405176737."],["epoch-gpqa-Hc2ycMP6DyCZxCPjKbZFQU","gemini-1-5-flash","Gemini 1.5 Flash (Sep '24)","Gemini 1.5 Flash (May 2024)","gemini-1-5-flash-epoch-gemini-1-5-flash-001","Gemini 1.5 Flash (May 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",40.372475,40.372475,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.4±0.02; stderr=0.023852475604024478."],["epoch-gpqa-PCLvrs4TEBtqQqcYeU8pAG","gemini-1-5-flash","Gemini 1.5 Flash (Sep '24)","gemini-1.5-flash-002","gemini-1-5-flash-epoch-gemini-1-5-flash-002","gemini-1.5-flash-002","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",47.316919,47.316919,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.47±0.03; stderr=0.027737408277040958."],["epoch-gpqa-gFT36W4KU3wggLQ9nNCctg","gemini-1-5-flash-8b","Gemini 1.5 Flash-8B","gemini-1.5-flash-8b-001","gemini-1-5-flash-8b-epoch-gemini-1-5-flash-8b-001","gemini-1.5-flash-8b-001","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",32.954545,32.954545,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-03-12","2025-03-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-03-12","2026-09-01","2026-08-18","source-checked","choice:0.33±0.02; stderr=0.02416551116211757."],["epoch-gpqa-PKdqpKZ9tqDWyWJ7GksBWz","gemini-1-5-pro","Gemini 1.5 Pro","gemini-1.5-pro-001","gemini-1-5-pro-epoch-gemini-1-5-pro-001","gemini-1.5-pro-001","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",45.864899,45.864899,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.46±0.03; stderr=0.02617295639558329."],["epoch-gpqa-9TrLEEw8TvsyugyCjwvUvD","gemini-1-5-pro","Gemini 1.5 Pro","gemini-1.5-pro-002","gemini-1-5-pro-epoch-gemini-1-5-pro-002","gemini-1.5-pro-002","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",57.228535,57.228535,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.57±0.03; stderr=0.027959709636504244."],["epoch-gpqa-fbszwTApm3f28z5VA6oXBK","gemini-2-0-flash","Gemini 2.0 Flash (Feb '25)","Gemini 2.0 Flash (Feb 2025)","gemini-2-0-flash-epoch-gemini-2-0-flash-001","Gemini 2.0 Flash (Feb 2025)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",64.141414,64.141414,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-06","2025-02-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-06","2026-09-01","2026-08-18","source-checked","choice:0.64±0.03; stderr=0.027395618179027574."],["epoch-gpqa-Q4qTzHnjLyiCCRvNpXioUq","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro Preview (Jun 2025)","gemini-2-5-pro-epoch-gemini-2-5-pro-preview-06-05","Gemini 2.5 Pro Preview (Jun 2025)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",84.848485,84.848485,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-06-05","2025-06-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-06-05","2026-09-01","2026-08-17","source-checked","choice:0.85±0.03; stderr=0.02554565042660364."],["epoch-gpqa-eZaQhpbhbzbo3sbKGS5Efa","gpt-35-turbo","GPT-3.5 Turbo","GPT-3.5 Turbo (Jan 2024)","gpt-35-turbo-epoch-gpt-3-5-turbo-0125","GPT-3.5 Turbo (Jan 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",27.17803,27.17803,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.27±0.02; stderr=0.01953579968788737."],["epoch-gpqa-enhRcQmfN6RFB2zu7R3Uxr","gpt-35-turbo","GPT-3.5 Turbo","GPT-3.5 Turbo (Nov 2023)","gpt-35-turbo-epoch-gpt-3-5-turbo-1106","GPT-3.5 Turbo (Nov 2023)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",28.030303,28.030303,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.28±0.02; stderr=0.018720372561004013."],["epoch-gpqa-Gj8pd5gU9r4LqwDK8XZ4mL","gpt-4-1","GPT-4.1","GPT-4.1","gpt-4-1-epoch-gpt-4-1-2025-04-14","GPT-4.1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",66.919192,66.919192,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-14","2025-04-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-14","2026-09-01","2026-08-18","source-checked","choice:0.67±0.03; stderr=0.02821440093855908."],["epoch-gpqa-dXHddeGRX7JRVa5UNfad5C","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",65.84596,65.84596,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-14","2025-04-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-14","2026-09-01","2026-08-18","source-checked","choice:0.66±0.03; stderr=0.027051500965295915."],["epoch-gpqa-azWfeFPP6zSocuQX3JYBU5","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",48.926768,48.926768,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-14","2025-04-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-14","2026-09-01","2026-08-18","source-checked","choice:0.49±0.02; stderr=0.024849734950435368."],["epoch-gpqa-dodibSv7pk8GtqeMy4YLQg","gpt-4-5","GPT-4.5 (Preview)","GPT-4.5 Preview (Feb 2025)","gpt-4-5-epoch-gpt-4-5-preview-2025-02-27","GPT-4.5 Preview (Feb 2025)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",68.686869,68.686869,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-28","2025-02-28","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-28","2026-09-01","2026-08-18","source-checked","choice:0.69±0.03; stderr=0.033042050878136546."],["epoch-gpqa-fairJpL9VekDBKxYHG9ZoG","gpt-4o","GPT-4o","GPT-4o (Aug 2024)","gpt-4o-epoch-gpt-4o-2024-08-06","GPT-4o (Aug 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",49.210859,49.210859,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.49±0.03; stderr=0.025999270277271707."],["epoch-gpqa-Y59q8WbEDp2HH8oovyJCj9","gpt-4o","GPT-4o","GPT-4o (May 2024)","gpt-4o-epoch-gpt-4o-2024-05-13","GPT-4o (May 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",48.895202,48.895202,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.49±0.03; stderr=0.026254109873371994."],["epoch-gpqa-gULKGqAe9nG9syxmYMtkhi","gpt-4o","GPT-4o","GPT-4o (Nov 2024)","gpt-4o-epoch-gpt-4o-2024-11-20","GPT-4o (Nov 2024)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",47.885101,47.885101,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-05","2025-02-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-05","2026-09-01","2026-08-18","source-checked","choice:0.48±0.03; stderr=0.026336972178726107."],["epoch-gpqa-L9yfwZC3snsKZshBXTaXjc","gpt-4o-mini","GPT-4o mini","gpt-4o-mini-2024-07-18","gpt-4o-mini-epoch-gpt-4o-mini-2024-07-18","gpt-4o-mini-2024-07-18","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",37.72096,37.72096,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.38±0.02; stderr=0.024068925424620136."],["epoch-gpqa-YQCwJRoGbGAqNCQoaL6cjk","gpt-5","GPT-5","GPT-5 (medium)","gpt-5-medium","GPT-5 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",85.353535,85.353535,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-07","2025-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-07","2026-09-01","2026-08-17","source-checked","choice:0.85±0.02; stderr=0.021314825297984473."],["epoch-gpqa-CFNzodoTSVVVgp9PUXftUP","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",71.65404,71.65404,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-07","2025-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-07","2026-09-01","2026-08-17","source-checked","choice:0.72±0.02; stderr=0.023192741878083397."],["epoch-gpqa-ELAxkvAigvEVdMmhxiaMiy","gpt-5-nano","GPT-5 nano","GPT-5 nano (medium)","gpt-5-nano-medium","GPT-5 nano (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",67.424242,67.424242,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-08-07","2025-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-08-07","2026-09-01","2026-08-18","source-checked","choice:0.67±0.03; stderr=0.026607011205052013."],["epoch-gpqa-FQ8Ykq4JWRSXnyF9rbmLc7","grok-3","Grok 3","Grok 3 (beta)","grok-3-epoch-grok-3-beta","Grok 3 (beta)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",75.757576,75.757576,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-26","2025-05-26","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-26","2026-09-01","2026-08-18","source-checked","choice:0.76±0.03,choice:0.59±0.02; stderr=0.026863078746340353."],["epoch-gpqa-JFZMWFVGNi5R5TdZrvQUFw","grok-3-mini","Grok 3 mini","grok-3-mini-beta_high","grok-3-mini-epoch-grok-3-mini-beta-high","grok-3-mini-beta_high","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",75.505051,75.505051,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-26","2025-05-26","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-26","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03,choice:0.76±0.02; stderr=0.027675521820041432."],["epoch-gpqa-McxfQdC5aypnD3STwZUBjH","grok-3-mini","Grok 3 mini","grok-3-mini-beta_low","grok-3-mini-epoch-grok-3-mini-beta-low","grok-3-mini-beta_low","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",76.262626,76.262626,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-10","2025-04-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-10","2026-09-01","2026-08-17","source-checked","choice:0.76±0.03; stderr=0.03031371053819892."],["epoch-gpqa-jrvutCoFFG4KxYtU7REdpe","llama-3-70b","Llama 3 70B","Meta-Llama-3-70B-Instruct","llama-3-70b-epoch-meta-llama-3-70b-instruct","Meta-Llama-3-70B-Instruct","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",40.561869,40.561869,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.41±0.03; stderr=0.026224849492112068."],["epoch-gpqa-S5QYXSvQBRSbUbXSnAGbMm","llama-3-1-405b","Llama 3.1 405B","Llama-3.1-405B-Instruct","llama-3-1-405b-epoch-llama-3-1-405b-instruct","Llama-3.1-405B-Instruct","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",50.915404,50.915404,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.51±0.03; stderr=0.025861992702770648."],["epoch-gpqa-LyHy99ubGCqBhoaksQjkiF","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (FP8)","llama-4-maverick-epoch-llama-4-maverick-17b-128e-instruct-fp8","Llama 4 Maverick (FP8)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",66.982323,66.982323,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-08","2025-04-08","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-08","2026-09-01","2026-08-18","source-checked","choice:0.67±0.03; stderr=0.028244823529728173."],["epoch-gpqa-MsXAXDEKP3xBKtiA7aTBBG","llama-4-scout","Llama 4 Scout","Llama-4-Scout-17B-16E-Instruct","llama-4-scout-epoch-llama-4-scout-17b-16e-instruct","Llama-4-Scout-17B-16E-Instruct","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",51.830808,51.830808,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-08","2025-04-08","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-08","2026-09-01","2026-08-18","source-checked","choice:0.52±0.03; stderr=0.031549020124994755."],["epoch-gpqa-dDeREdDJ5bVyci3APZxzWb","mistral-large","Mistral Large (Feb '24)","mistral-large-2402","mistral-large-epoch-mistral-large-2402","mistral-large-2402","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",38.762626,38.762626,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.39±0.02; stderr=0.024776598535417284."],["epoch-gpqa-ZQPNnicTuhuscJvni85PFY","mistral-large-2","Mistral Large 2","mistral-large-2407","mistral-large-2-epoch-mistral-large-2407","mistral-large-2407","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",49.021465,49.021465,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.49±0.03; stderr=0.025918840988656967."],["epoch-gpqa-n9gpXfUzo8SbWqvrG2q4xM","mistral-large-2","Mistral Large 2","mistral-large-2411","mistral-large-2-epoch-mistral-large-2411","mistral-large-2411","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",51.325758,51.325758,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-25","2025-02-25","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-25","2026-09-01","2026-08-18","source-checked","choice:0.51±0.03; stderr=0.026887818295509087."],["epoch-gpqa-FNMbfvEEm9PRnrJnrqk2PH","mistral-medium-3","Mistral Medium 3","mistral-medium-2505","mistral-medium-3-epoch-mistral-medium-2505","mistral-medium-2505","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",59.532828,59.532828,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-05-07","2025-05-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-05-07","2026-09-01","2026-08-18","source-checked","choice:0.6±0.03; stderr=0.02824697229654645."],["epoch-gpqa-oDq358GfyV5Au65scERdVo","mistral-small-3","Mistral Small 3","mistral-small-2501","mistral-small-3-epoch-mistral-small-2501","mistral-small-2501","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",45.296717,45.296717,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-30","2025-01-30","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-30","2026-09-01","2026-08-18","source-checked","choice:0.45±0.02; stderr=0.024794688677057853."],["epoch-gpqa-S8bBDAbucsFceirHxeEDvu","mistral-small-3-1","Mistral Small 3.1","mistral-small-2503","mistral-small-3-1-epoch-mistral-small-2503","mistral-small-2503","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",47.474747,47.474747,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-03-18","2025-03-18","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-03-18","2026-09-01","2026-08-18","source-checked","choice:0.47±0.03; stderr=0.027616495140012243."],["epoch-gpqa-NAkCFiFiDSN3NXMj7HMvMS","o1","o1","o1 (high)","o1-epoch-o1-2024-12-17-high","o1 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",76.767677,76.767677,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-13","2025-02-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-13","2026-09-01","2026-08-17","source-checked","choice:0.77±0.03; stderr=0.030088629490217445."],["epoch-gpqa-Uczhz7MKLstjLkSbbGuwdN","o1","o1","o1 (medium)","o1-epoch-o1-2024-12-17-medium","o1 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",75.757576,75.757576,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-17","source-checked","choice:0.76±0.03; stderr=0.030532892233932022."],["epoch-gpqa-X2gKewiGPZx5DLFEddUUqW","o1-mini","o1-mini","o1-mini (high)","o1-mini-epoch-o1-mini-2024-09-12-high","o1-mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",62.373737,62.373737,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-13","2025-02-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-13","2026-09-01","2026-08-18","source-checked","choice:0.62±0.03; stderr=0.027427723511230753."],["epoch-gpqa-4NiUpCrLfz3oFaerkrMeBt","o1-mini","o1-mini","o1-mini (medium)","o1-mini-epoch-o1-mini-2024-09-12-medium","o1-mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",59.501263,59.501263,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.6±0.03; stderr=0.02795124238637276."],["epoch-gpqa-7CkADTsoNbDd88uqYbqfwj","o1-preview","o1-preview","o1-preview","o1-preview-epoch-o1-preview-2024-09-12","o1-preview","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",50.315657,50.315657,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.5±0.03; stderr=0.028257821072909393."],["epoch-gpqa-evANZQ9oQTdGYbDERw7VrD","o3","o3","o3 (high)","o3-epoch-o3-2025-04-16-high","o3 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",81.818182,81.818182,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-16","2025-04-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-16","2026-09-01","2026-08-17","source-checked","choice:0.82±0.02; stderr=0.021267123846387202."],["epoch-gpqa-MzsTvCEEfEpzn4WvxYdJkX","o3-mini","o3-mini","o3-mini (high)","o3-mini-high","o3-mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",77.020202,77.020202,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-02-13","2025-02-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-02-13","2026-09-01","2026-08-18","source-checked","choice:0.77±0.03; stderr=0.025855332698878485."],["epoch-gpqa-ehvA5nisC7GMbgbyy3Z4Et","o3-mini","o3-mini","o3-mini (medium)","o3-mini-epoch-o3-mini-2025-01-31-medium","o3-mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",74.27399,74.27399,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-31","2025-01-31","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-31","2026-09-01","2026-08-18","source-checked","choice:0.74±0.03; stderr=0.025411943362857046."],["epoch-gpqa-9Sgsz5XCjscuZkexf7MX9c","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",79.608586,79.608586,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-16","2025-04-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-16","2026-09-01","2026-08-17","source-checked","choice:0.8±0.02; stderr=0.024034764096121958."],["epoch-gpqa-RAwuPRwJedy7P2KeQhEmr8","phi-4","Phi-4","phi-4","phi-4-epoch-phi-4","phi-4","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",56.060606,56.060606,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-31","2025-01-31","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-31","2026-09-01","2026-08-18","source-checked","choice:0.56±0.03; stderr=0.02590274664132679."],["epoch-gpqa-GBf7rqHkGGzfX87V37F9mH","qwen-2-5-max","Qwen2.5 Max","Qwen2.5-Max","qwen-2-5-max-epoch-qwen-max-2025-01-25","Qwen2.5-Max","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",56.123737,56.123737,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-01","2025-04-01","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-01","2026-09-01","2026-08-18","source-checked","choice:0.56±0.03; stderr=0.027858831475709448."],["epoch-gpqa-9Ke5At9bYuf56SmKJJ7Sz7","qwen-turbo","Qwen2.5 Turbo","Qwen Turbo","qwen-turbo-epoch-qwen-turbo-2024-11-01","Qwen Turbo","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",41.792929,41.792929,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-04-07","2025-04-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-04-07","2026-09-01","2026-08-18","source-checked","choice:0.42±0.03; stderr=0.02617727508870251."],["epoch-gpqa-WtALgW6VSFdihDZWXRm2AZ","qwen2-5-72b","Qwen2.5-72B","Qwen2.5-72B","qwen2-5-72b-epoch-qwen2-5-72b-instruct","Qwen2.5-72B","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",49.147727,49.147727,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-01-27","2025-01-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-01-27","2026-09-01","2026-08-18","source-checked","choice:0.49±0.03; stderr=0.026947767404178667."],["epoch-gpqa-46RCssznDADYeg8MuSgvsj","qwen3-max","Qwen3 Max","Qwen3 Max","qwen3-max-epoch-qwen3-max-2025-09-23","Qwen3 Max","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.0",72.60101,72.60101,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-06","2025-10-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-06","2026-09-01","2026-08-17","source-checked","choice:0.73±0.03; stderr=0.02734528103171235."],["epoch-gpqa-cqUZ6gcq2HBnShFRWvKjGr","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro","gemini-2-5-pro-epoch-gemini-2-5-pro","Gemini 2.5 Pro","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",85.290404,85.290404,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-16","2025-11-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-16","2026-09-01","2026-08-17","source-checked","choice:0.85±0.02; stderr=0.02117010842348584."],["epoch-gpqa-9nPyNrZxwtoT7eS4DZbQF6","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",86.174242,86.174242,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-10-29","2025-10-29","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-29","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.020787836099851995."],["epoch-gpqa-jP8EyLtPFsakgnsHe2oJM6","gpt-5-mini","GPT-5 mini","GPT-5 mini (high)","gpt-5-mini-high","GPT-5 mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",75,75,"percent","higher","2.2.0","ranking-eligible","direct","2025-10-30","2025-10-30","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-30","2026-09-01","2026-08-17","source-checked","choice:0.75±0.02; stderr=0.022767295427113803."],["epoch-gpqa-2pSDjMSGDh4HsAxkqUuNQP","gpt-5-nano","GPT-5 nano","GPT-5 nano (high)","gpt-5-nano-high","GPT-5 nano (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",69.444444,69.444444,"percent","higher","2.2.0","ranking-eligible","direct","2025-10-30","2025-10-30","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-10-30","2026-09-01","2026-08-18","source-checked","choice:0.69±0.03; stderr=0.027594508998251045."],["epoch-gpqa-CSVtMksEVjBNsGJGjW4GcQ","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",87.626263,87.626263,"percent","higher","2.2.0","ranking-eligible","direct","2025-11-13","2025-11-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-13","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.01875368281077595."],["epoch-gpqa-V3YQtSEnQ3TPZEPPgkCKQM","gpt-5-1","GPT-5.1","GPT-5.1 (medium)","gpt-5-1-epoch-gpt-5-1-2025-11-13-medium","GPT-5.1 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",85.037879,85.037879,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-17","2025-11-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-17","2026-09-01","2026-08-17","source-checked","choice:0.85±0.02; stderr=0.021250422082249816."],["epoch-gpqa-muGecFpAg9wHd6gexWJ99L","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking Turbo","kimi-k2-thinking-epoch-kimi-k2-thinking-turbo","Kimi K2 Thinking Turbo","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.1",84.217172,84.217172,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-11","2025-11-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-11","2026-09-01","2026-08-17","source-checked","choice:0.84±0.02; stderr=0.020879845731985747."],["epoch-gpqa-dqgqP2uEPswfECCVj2Kbr9","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max)","claude-opus-4-8-max","Claude Opus 4.8 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.10",91.035354,91.035354,"percent","higher","2.2.0","ranking-eligible","direct","2026-06-07","2026-06-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-07","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.019157125794612026."],["epoch-gpqa-d23hYvSNGFo7rniLFyBWiu","claude-fable-5","Claude Fable 5","Claude Fable 5 (high)","claude-fable-5-high","Claude Fable 5 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",83.333333,83.333333,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.83±0.03; stderr=0.026552207828215258."],["epoch-gpqa-U5PsUdX2ak4kJ2hgnQkWnX","claude-fable-5","Claude Fable 5","Claude Fable 5 (low)","claude-fable-5-epoch-claude-fable-5-low","Claude Fable 5 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",78.787879,78.787879,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.79±0.03; stderr=0.029126522834586777."],["epoch-gpqa-S5pm7ctk9zgPfbyHrbD9AV","claude-fable-5","Claude Fable 5","Claude Fable 5 (max)","claude-fable-5-max","Claude Fable 5 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",85.858586,85.858586,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.024825909793343353."],["epoch-gpqa-JcQMgXdZKU54u52B8TGQKp","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (max)","claude-opus-4-6-max","Claude Opus 4.6 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.383838,88.383838,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.02282888177524936."],["epoch-gpqa-Ju7pUkkwYVj6ugi9Ri5Jjp","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (max)","claude-opus-4-7-max","Claude Opus 4.7 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",86.363636,86.363636,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.02445015597318985."],["epoch-gpqa-GThtce6GAmRAS8srQyGQoA","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (low)","claude-opus-4-8-epoch-claude-opus-4-8-low","Claude Opus 4.8 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.383838,88.383838,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.022828881775249357."],["epoch-gpqa-ckbZdkrXPwbFdjn6bmUQsi","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (no thinking)","claude-opus-4-8-epoch-claude-opus-4-8-none","Claude Opus 4.8 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",85.353535,85.353535,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.85±0.03; stderr=0.025190921114603887."],["epoch-gpqa-mLmUnBuX6DYKtb6x46GteM","claude-opus-5","Claude Opus 5","Claude Opus 5","claude-opus-5-epoch-claude-opus-5","Claude Opus 5","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",92.929293,92.929293,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.018263105420199485."],["epoch-gpqa-WfaJS4L7Xdz5ZBXgJMbRGb","claude-opus-5","Claude Opus 5","Claude Opus 5 (low)","claude-opus-5-low","Claude Opus 5 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",87.878788,87.878788,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.02325315795194205."],["epoch-gpqa-D3mtasTumfvjR4fjmnHWRy","claude-opus-5","Claude Opus 5","Claude Opus 5 (max)","claude-opus-5-max","Claude Opus 5 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.876263,93.876263,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-24","2026-07-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-24","2026-09-01","2026-08-17","source-checked","choice:0.94±0.01; stderr=0.014782015056089346."],["epoch-gpqa-DmdKPD9Z9RXwYyATAoG4Uw","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (high)","claude-sonnet-4-6-non-reasoning-high","Claude Sonnet 4.6 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",83.333333,83.333333,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","choice:0.83±0.03; stderr=0.026552207828215258."],["epoch-gpqa-SEePRKRcv8oqn5HpD2KFxB","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (max)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",78.787879,78.787879,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.79±0.03; stderr=0.029126522834586777."],["epoch-gpqa-e8MXnAM8HMF8eEzYtYA7xv","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (medium)","claude-sonnet-4-6-livebench-2026-06-25-medium","Claude Sonnet 4.6 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",83.333333,83.333333,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","choice:0.83±0.03; stderr=0.026552207828215258."],["epoch-gpqa-FNXAt76ZUS3sWhQDFMga7U","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max)","claude-sonnet-5-max","Claude Sonnet 5 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",80.30303,80.30303,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.8±0.03; stderr=0.028335609732463334."],["epoch-gpqa-HhiYxEx3Bf8YDPUh34K4DC","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (xhigh)","claude-sonnet-5-xhigh","Claude Sonnet 5 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",90.530303,90.530303,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-01","2026-07-01","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-01","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.01783832131341021."],["epoch-gpqa-6osoAGM4euiAAksriYyLtj","deepseek-v3-2","DeepSeek V3.2","deepseek-chat","deepseek-v3-2-epoch-deepseek-chat","deepseek-chat","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",71.212121,71.212121,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-16","2026-07-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-16","2026-09-01","2026-08-17","source-checked","choice:0.71±0.03; stderr=0.03225883512300997."],["epoch-gpqa-EDFusDaaFkh5n2MqKNxMBD","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek V4 Flash 0731 (max)","deepseek-v4-flash-0731-epoch-deepseek-v4-flash-0731-max","DeepSeek V4 Flash 0731 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",91.035354,91.035354,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.017655466909172534."],["epoch-gpqa-TPdRwHTJtaJYSu2Cxc5Czn","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek v4 (high)","deepseek-v4-pro-high","DeepSeek v4 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",90.909091,90.909091,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","choice:0.91±0.02; stderr=0.020482086775424225."],["epoch-gpqa-LuZUnhD5rBLjRw36DaqfvZ","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek v4 (max)","deepseek-v4-pro-max","DeepSeek v4 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.646465,89.646465,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-16","2026-06-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-16","2026-09-01","2026-08-18","source-checked","choice:0.9±0.02; stderr=0.017473014662966007."],["epoch-gpqa-FtGueFbnFnKBAuLmEEufFg","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek v4 (no thinking)","deepseek-v4-pro-epoch-deepseek-v4-pro-none","DeepSeek v4 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",73.232323,73.232323,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","choice:0.73±0.03; stderr=0.031544498882702825."],["epoch-gpqa-gVdraDT3pXmaExP9VbqME4","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (high)","gemini-3-flash-epoch-gemini-3-flash-preview-high","Gemini 3 Flash Preview (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.393939,89.393939,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.021938047738853106."],["epoch-gpqa-nE4hmJo3QV6X5jeGWAeAze","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite (high)","gemini-3-1-flash-lite-epoch-gemini-3-1-flash-lite-high","Gemini 3.1 Flash-Lite (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",81.818182,81.818182,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.82±0.03; stderr=0.027479603010538825."],["epoch-gpqa-99ys2jgQiPQ5UwrqnazfqA","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite (low)","gemini-3-1-flash-lite-epoch-gemini-3-1-flash-lite-low","Gemini 3.1 Flash-Lite (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",74.242424,74.242424,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03; stderr=0.03115626951964683."],["epoch-gpqa-CmHy8cVECGkvQTwakfjRo2","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite (minimal)","gemini-3-1-flash-lite-epoch-gemini-3-1-flash-lite-minimal","Gemini 3.1 Flash-Lite (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",73.737374,73.737374,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03; stderr=0.031353050095330834."],["epoch-gpqa-BgmrBFsbsaD8rNpebv49Da","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview (high)","gemini-3-1-pro-preview-livebench-2026-06-25-high","Gemini 3.1 Pro Preview (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",94.444444,94.444444,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.94±0.02; stderr=0.016319950700767385."],["epoch-gpqa-eiTYduCDNq3HgwTg4ZhEw6","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (low)","gemini-3-5-flash-epoch-gemini-3-5-flash-low","Gemini 3.5 Flash (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.888889,88.888889,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.02239078763821682."],["epoch-gpqa-QDZRoAVaEDtdS6nECEeE35","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (minimal)","gemini-3-5-flash-epoch-gemini-3-5-flash-minimal","Gemini 3.5 Flash (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",86.363636,86.363636,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.02445015597318985."],["epoch-gpqa-chbh6XstiqtUa5W9zuyfjS","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Gemini 3.5 Flash-Lite (high)","gemini-3-5-flash-lite-livebench-2026-06-25-high","Gemini 3.5 Flash-Lite (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",83.333333,83.333333,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.83±0.03; stderr=0.026552207828215258."],["epoch-gpqa-dgSJTFQ4UtuQKEhB9ss6K7","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Gemini 3.5 Flash-Lite (low)","gemini-3-5-flash-lite-epoch-gemini-3-5-flash-lite-low","Gemini 3.5 Flash-Lite (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",75.757576,75.757576,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.76±0.03; stderr=0.030532892233932022."],["epoch-gpqa-7aAKcnsCXoSumfAu8mr2un","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Gemini 3.5 Flash-Lite (minimal)","gemini-3-5-flash-lite-epoch-gemini-3-5-flash-lite-minimal","Gemini 3.5 Flash-Lite (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",74.242424,74.242424,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03; stderr=0.031156269519646826."],["epoch-gpqa-86SY5JbPB49xXhHFQW6tHu","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high)","gemini-3-6-flash-high","Gemini 3.6 Flash (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",94.128788,94.128788,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-02","2026-08-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-02","2026-09-01","2026-08-17","source-checked","choice:0.94±0.01; stderr=0.014028967942064968."],["epoch-gpqa-JoraLb9y55QXpMg4Rrmr5x","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (low)","gemini-3-6-flash-epoch-gemini-3-6-flash-low","Gemini 3.6 Flash (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",86.363636,86.363636,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.02445015597318985."],["epoch-gpqa-aPmR8fV49N98a6NVKpASAf","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (minimal)","gemini-3-6-flash-epoch-gemini-3-6-flash-minimal","Gemini 3.6 Flash (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",85.858586,85.858586,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.024825909793343353."],["epoch-gpqa-XyVL3VffJ6UboY97vxpohL","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high)","gemini-3-7-flash-high","Gemini 3.7 Flash (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",94.823232,94.823232,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","choice:0.95±0.01; stderr=0.013456525292246259."],["epoch-gpqa-AsVZCdJLQvNykhKjisihe2","gemma-3-27b","Gemma 3 27B","gemma-3-27b-it","gemma-3-27b-epoch-gemma-3-27b-it","gemma-3-27b-it","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",43.939394,43.939394,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","choice:0.44±0.04; stderr=0.03536085947529479."],["epoch-gpqa-P5L8qbXo42vTvbU3UvTNjk","gemma-4-31b","Gemma 4 31B","Gemma 4 31B IT (minimal)","gemma-4-31b-epoch-gemma-4-31b-it-minimal","Gemma 4 31B IT (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",75.757576,75.757576,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-17","source-checked","choice:0.76±0.03; stderr=0.03053289223393202."],["epoch-gpqa-CBbeMKRRSeRsXnvSfn7Png","glm-5-1","GLM-5.1","GLM-5.1","glm-5-1-epoch-glm-5-1","GLM-5.1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.89899,89.89899,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-10","2026-08-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-10","2026-09-01","2026-08-17","source-checked","choice:0.9±0.02; stderr=0.021469735576055356."],["epoch-gpqa-c5kpB4axzmygRywWqBKFmv","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",91.856061,91.856061,"percent","higher","2.2.0","ranking-eligible","direct","2026-06-24","2026-06-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-24","2026-09-01","2026-08-17","source-checked","choice:0.92±0.02; stderr=0.016123207533687733."],["epoch-gpqa-YDQJAMgMjLaikPkeVDbAjr","glm-5-2","GLM-5.2","GLM-5.2 (no thinking)","glm-5-2-epoch-glm-5-2-none","GLM-5.2 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",71.212121,71.212121,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-10","2026-08-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-10","2026-09-01","2026-08-17","source-checked","choice:0.71±0.03; stderr=0.03225883512300997."],["epoch-gpqa-dLDBahYBTGRNjbsXA5pyr3","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",90.909091,90.909091,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-24","2026-08-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-24","2026-09-01","2026-09-01","independently-verified","choice:0.91±0.02; stderr=0.015851578202142853; Epoch run dLDBahYBTGRNjbsXA5pyr3. Epoch AI independently evaluated the exact canonical default configuration on the current GPQA Diamond protocol."],["epoch-gpqa-n4wyt7wHjhDdjrGacE9smk","gpt-5","GPT-5","GPT-5 (minimal)","gpt-5-epoch-gpt-5-2025-08-07-minimal","GPT-5 (minimal)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",71.717172,71.717172,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-20","2026-07-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-20","2026-09-01","2026-08-17","source-checked","choice:0.72±0.03; stderr=0.032087795587867514."],["epoch-gpqa-hYSstHWNUNwYoZdmZpYQqK","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",71.717172,71.717172,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.72±0.03; stderr=0.032087795587867514."],["epoch-gpqa-hp4x7bhBYhkQqZoZFf4dTZ","gpt-5-nano","GPT-5 nano","GPT-5 nano (low)","gpt-5-nano-epoch-gpt-5-nano-2025-08-07-minimal","GPT-5 nano (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",48.484848,48.484848,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-18","source-checked","choice:0.48±0.04; stderr=0.03560716516531066."],["epoch-gpqa-mNZ34pXP8JN779JoHnjNQc","gpt-5-nano","GPT-5 nano","GPT-5 nano (low)","gpt-5-nano-epoch-gpt-5-nano-2025-08-07-low","GPT-5 nano (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",57.575758,57.575758,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-18","source-checked","choice:0.58±0.04; stderr=0.03521224908841589."],["epoch-gpqa-UT5fVL4EGvSTZBKMTPy9YN","gpt-5-1","GPT-5.1","GPT-5.1 (no thinking)","gpt-5-1-epoch-gpt-5-1-2025-11-13-none","GPT-5.1 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",66.666667,66.666667,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.67±0.03; stderr=0.033586181457325226."],["epoch-gpqa-4sPs53wmM9bvN8JXzsiNCk","gpt-5-2","GPT-5.2","GPT-5.2 (none)","gpt-5-2-epoch-gpt-5-2-2025-12-11-none","GPT-5.2 (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",73.232323,73.232323,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","choice:0.73±0.03; stderr=0.031544498882702825."],["epoch-gpqa-FbhcT8NnJUntcyNGubJXY6","gpt-5-4","GPT-5.4","GPT-5.4 (high)","gpt-5-4-epoch-gpt-5-4-2026-03-05-high","GPT-5.4 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.89899,89.89899,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.9±0.02; stderr=0.021469735576055356."],["epoch-gpqa-aYgDc2iSDpmHkQfu7YjCHc","gpt-5-4","GPT-5.4","GPT-5.4 (low)","gpt-5-4-low","GPT-5.4 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",84.848485,84.848485,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.85±0.03; stderr=0.025545650426603637."],["epoch-gpqa-ManJk43Kf9VuXowNqoDrov","gpt-5-4","GPT-5.4","GPT-5.4 (medium)","gpt-5-4-epoch-gpt-5-4-2026-03-05-medium","GPT-5.4 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.888889,88.888889,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.02239078763821682."],["epoch-gpqa-LaBuzZExRqocD4FET6yx26","gpt-5-4","GPT-5.4","GPT-5.4 (none)","gpt-5-4-epoch-gpt-5-4-2026-03-05-none","GPT-5.4 (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",74.747475,74.747475,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.75±0.03; stderr=0.030954055470365872."],["epoch-gpqa-Qyz2Qt5x4P2XwB3w4h4jFR","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (none)","gpt-5-4-mini-epoch-gpt-5-4-mini-2026-03-17-none","GPT-5.4 mini (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",64.141414,64.141414,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.64±0.03; stderr=0.034169036403915276."],["epoch-gpqa-gKB2rCD7mHzFN2MCt2YCni","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",86.868687,86.868687,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.87±0.02; stderr=0.02406315641682249."],["epoch-gpqa-Nq7w5iKALDSuT5oQdn5m8f","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (low)","gpt-5-4-nano-epoch-gpt-5-4-nano-2026-03-17-low","GPT-5.4 nano (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",72.222222,72.222222,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.72±0.03; stderr=0.03191178226713548."],["epoch-gpqa-gLewKMc7ttCD2moUrgAyYX","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (no thinking)","gpt-5-4-nano-epoch-gpt-5-4-nano-2026-03-17-none","GPT-5.4 nano (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",55.555556,55.555556,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.56±0.04; stderr=0.035402943770953675."],["epoch-gpqa-KesKa5EXtFG7H4UMBFzgCh","gpt-5-5","GPT-5.5","GPT-5.5 (no thinking)","gpt-5-5-epoch-gpt-5-5-none","GPT-5.5 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",77.272727,77.272727,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.77±0.03; stderr=0.029857515673386438."],["epoch-gpqa-nW39ZzuDQ3xTAGVoEfFG95","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (low)","gpt-5-6-luna-low","GPT-5.6 Luna (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",82.323232,82.323232,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.82±0.03; stderr=0.027178752639044908."],["epoch-gpqa-LWkDtkmd3TNmLzMhfZD5HW","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",91.603535,91.603535,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","choice:0.92±0.02; stderr=0.01728284355494035."],["epoch-gpqa-bDKKx7zTdW5WwMCfwqsLi4","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (none)","gpt-5-6-luna-epoch-gpt-5-6-luna-none","GPT-5.6 Luna (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",63.636364,63.636364,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.64±0.03; stderr=0.03427308652999934."],["epoch-gpqa-6cmgkNAQZekBr6e6rVdLe9","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (low)","gpt-5-6-sol-low","GPT-5.6 Sol (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.89899,89.89899,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.9±0.02; stderr=0.021469735576055356."],["epoch-gpqa-nuKCgZReCBsXKCkNj9bK37","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.497475,93.497475,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.01570109070867593."],["epoch-gpqa-DDfYPyKQW7vgDtGgkwcKSi","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (none)","gpt-5-6-sol-epoch-gpt-5-6-sol-none","GPT-5.6 Sol (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",82.828283,82.828283,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.83±0.03; stderr=0.026869716187429872."],["epoch-gpqa-JcuoAWhcnngPEZBPmiUyLF","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (low)","gpt-5-6-terra-low","GPT-5.6 Terra (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",87.373737,87.373737,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.87±0.02; stderr=0.023664359402880177."],["epoch-gpqa-4bLcFNu5zF6YuZf3qVKoHE","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.308081,93.308081,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-09","2026-07-09","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-09","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.015363672147634115."],["epoch-gpqa-F7v3JtKFb27j2YVmsndLM5","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (none)","gpt-5-6-terra-epoch-gpt-5-6-terra-none","GPT-5.6 Terra (none)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",77.272727,77.272727,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.77±0.03; stderr=0.029857515673386438."],["epoch-gpqa-hYATFPeUnMq7Std2nLUVAJ","gpt-oss-20b","GPT-OSS 20B","gpt-oss-20b_high","gpt-oss-20b-epoch-gpt-oss-20b-high","gpt-oss-20b_high","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",45.959596,45.959596,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-06","2026-08-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-06","2026-09-01","2026-08-18","source-checked","choice:0.46±0.04; stderr=0.03550702465131341."],["epoch-gpqa-STxFZiSPAPScUuENvXBWLc","grok-4-20","Grok 4.20","grok-4.20-0309-reasoning","grok-4-20-epoch-grok-4-20-0309-reasoning","grok-4.20-0309-reasoning","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",89.330808,89.330808,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-13","2026-07-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-13","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.01863338951496632."],["epoch-gpqa-Vw7spGUnJyf3vuzSjTcj2F","grok-4-3","Grok 4.3","grok-4.3_high","grok-4-3-epoch-grok-4-3-high","grok-4.3_high","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.825758,88.825758,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-06-17","2026-06-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-06-17","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.01963840420048217."],["epoch-gpqa-9kSgmUd9vh4CBXz3tJhctN","grok-4-5","Grok 4.5","Grok 4.5 (high)","grok-4-5-aa-2-high","Grok 4.5 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.434343,93.434343,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-08","2026-07-08","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-08","2026-09-01","2026-08-17","source-checked","choice:0.93±0.01; stderr=0.014340395485738813."],["epoch-gpqa-hnP6Z88x8bPEBRZ2PW6kps","grok-4-6","Grok 4.6","Grok 4.6 (high)","grok-4-6-high","Grok 4.6 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",94.002525,94.002525,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-12","2026-08-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-12","2026-09-01","2026-08-17","source-checked","choice:0.94±0.01; stderr=0.014480018812911388."],["epoch-gpqa-9YJf2nviwjM6WVhw8qeJta","grok-4-6","Grok 4.6","Grok 4.6 (xhigh)","grok-4-6-xhigh","Grok 4.6 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.181818,93.181818,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.015204299687218694."],["epoch-gpqa-dfmiZmjAcNkLuAJL4z65eb","inkling","Inkling","Inkling (xhigh)","inkling-xhigh","Inkling (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.257576,88.257576,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-05","2026-08-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-05","2026-09-01","2026-08-18","source-checked","choice:0.88±0.02; stderr=0.019355903133033585."],["epoch-gpqa-jTHw5SET8usd7emYPjEDjS","inkling-small","Inkling-Small","Inkling Small (xhigh)","inkling-small-epoch-inkling-small-xhigh","Inkling Small (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",88.510101,88.510101,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-14","2026-08-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-14","2026-09-01","2026-08-18","source-checked","choice:0.89±0.02; stderr=0.019100012831823398."],["epoch-gpqa-XneEig6pACkAWeekvgCwNk","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code","kimi-k2-7-code-epoch-kimi-k2-7-code","Kimi K2.7 Code","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",87.878788,87.878788,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.02325315795194205."],["epoch-gpqa-AoAD5uVtgVDnXmwA7STuno","kimi-k3","Kimi K3","Kimi K3 (high)","kimi-k3-epoch-kimi-k3-high","Kimi K3 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",91.919192,91.919192,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.92±0.02; stderr=0.019417681889724515."],["epoch-gpqa-mvNb3hNgSZ2gS6jHMtvCdk","kimi-k3","Kimi K3","Kimi K3 (low)","kimi-k3-low","Kimi K3 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",84.848485,84.848485,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.85±0.03; stderr=0.025545650426603637."],["epoch-gpqa-B8MaPUe3EYwUZASsNqE7C9","kimi-k3","Kimi K3","Kimi K3 (Max)","kimi-k3-max","Kimi K3 (Max)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",93.118687,93.118687,"percent","higher","2.2.0","ranking-eligible","direct","2026-07-16","2026-07-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-16","2026-09-01","2026-08-17","source-checked","choice:0.93±0.01; stderr=0.014937225364194858."],["epoch-gpqa-hDESmyGFrbZDomjff3rqzD","minimax-m3","MiniMax M3","MiniMax-M3","minimax-m3-epoch-minimax-m3","MiniMax-M3","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",90.909091,90.909091,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-10","2026-08-10","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-10","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.020482086775424225."],["epoch-gpqa-hP6P5e9zuHPToaQTTQC9A2","o1","o1","o1 (low)","o1-epoch-o1-2024-12-17-low","o1 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",74.242424,74.242424,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-20","2026-07-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-20","2026-09-01","2026-08-17","source-checked","choice:0.74±0.03; stderr=0.031156269519646826."],["epoch-gpqa-AZf6N4kygJGHPsSUkYtCba","o3","o3","o3 (low)","o3-epoch-o3-2025-04-16-low","o3 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",79.79798,79.79798,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-15","2026-07-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-15","2026-09-01","2026-08-17","source-checked","choice:0.8±0.03; stderr=0.028606204289229834."],["epoch-gpqa-TBYA7JWneLUCy9kyktFBbt","o3","o3","o3 (medium)","o3-epoch-o3-2025-04-16-medium","o3 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",80.808081,80.808081,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.81±0.03; stderr=0.028057791672989062."],["epoch-gpqa-hJf7Hxg4tgdftQ2QDNHMJA","o3-mini","o3-mini","o3-mini-2025-01-31_low","o3-mini-epoch-o3-mini-2025-01-31-low","o3-mini-2025-01-31_low","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",68.181818,68.181818,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-20","2026-07-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-20","2026-09-01","2026-08-18","source-checked","choice:0.68±0.03; stderr=0.0331847733384533."],["epoch-gpqa-XwRFKnuHv5qdUp9m5uTAEb","o4-mini","o4-mini","o4-mini (low)","o4-mini-epoch-o4-mini-2025-04-16-low","o4-mini (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",75.252525,75.252525,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-07-11","2026-07-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-07-11","2026-09-01","2026-08-17","source-checked","choice:0.75±0.03; stderr=0.030746300742124484."],["epoch-gpqa-noAGevebNp5W5yAnmbEYaz","o4-mini","o4-mini","o4-mini (medium)","o4-mini-epoch-o4-mini-2025-04-16-medium","o4-mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",77.777778,77.777778,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.78±0.03; stderr=0.02962022787479049."],["epoch-gpqa-bXrmQqzF9sXbuZfRrSc7NP","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B-A17B (no thinking)","qwen3-5-397b-epoch-qwen3-5-397b-a17b-none","Qwen3.5 397B-A17B (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",86.363636,86.363636,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.02445015597318985."],["epoch-gpqa-gZMFRbVKnBmcnZU4YqqKMN","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B-A17B (thinking)","qwen3-5-397b-epoch-qwen3-5-397b-a17b","Qwen3.5 397B-A17B (thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",85.858586,85.858586,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.024825909793343353."],["epoch-gpqa-WeP7Q27N5PzxSgg4ybgijX","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (no thinking)","qwen3-6-27b-epoch-qwen3-6-27b-none","Qwen3.6 27B (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",84.848485,84.848485,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.85±0.03; stderr=0.02554565042660364."],["epoch-gpqa-buLf8Ne29xmzMZBRQkvrQG","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (thinking)","qwen3-6-27b-epoch-qwen3-6-27b","Qwen3.6 27B (thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",85.858586,85.858586,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.024825909793343353."],["epoch-gpqa-Jv2hJvjpvA8xJW3Bzv7bfw","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview (thinking)","qwen3-6-max-epoch-qwen3-6-max-preview","Qwen3.6 Max Preview (thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",87.373737,87.373737,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.87±0.02; stderr=0.023664359402880177."],["epoch-gpqa-Yhds8GYkCXWgRQhuLiURua","qwen3-7-flash","Qwen3.7 Flash","Qwen3.7 Flash (no thinking)","qwen3-7-flash-epoch-qwen3-7-flash-none","Qwen3.7 Flash (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",80.808081,80.808081,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-18","source-checked","choice:0.81±0.03; stderr=0.028057791672989062."],["epoch-gpqa-i6i3gQf3ng2aAipvowPpws","qwen3-7-flash","Qwen3.7 Flash","Qwen3.7 Flash (thinking)","qwen3-7-flash-epoch-qwen3-7-flash","Qwen3.7 Flash (thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",82.323232,82.323232,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-18","source-checked","choice:0.82±0.03; stderr=0.027178752639044908."],["epoch-gpqa-W6mTanyjvXhasFBzHukv4S","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max","qwen-3-7-max-epoch-qwen3-7-max","Qwen3.7 Max","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",90.909091,90.909091,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.020482086775424225."],["epoch-gpqa-BWz8uig4QyqPtke59pM88z","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus (no thinking)","qwen-3-7-plus-epoch-qwen3-7-plus-none","Qwen3.7 Plus (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",81.818182,81.818182,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.82±0.03; stderr=0.027479603010538825."],["epoch-gpqa-URoWjw9Htc6QPwABVcrybw","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus (thinking)","qwen-3-7-plus-epoch-qwen3-7-plus","Qwen3.7 Plus (thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",87.878788,87.878788,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-08-07","2026-08-07","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-07","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.02325315795194205."],["epoch-gpqa-TEVNDC9QXBh7PQvX9kHZip","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.11",92.676768,92.676768,"percent","higher","2.2.0","ranking-eligible","direct","2026-08-04","2026-08-04","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-04","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.01686498021450146."],["epoch-gpqa-cXkPywkhy3NnW992ri73yG","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (16k thinking)","claude-opus-4-5-epoch-claude-opus-4-5-20251101-16k","Claude Opus 4.5 (16k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.2",85.479798,85.479798,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-25","2025-11-25","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-25","2026-09-01","2026-08-17","source-checked","choice:0.85±0.02; stderr=0.021508147720723254."],["epoch-gpqa-Kwwo3UFcPDk6iLAeSeeRaa","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (32k thinking)","claude-opus-4-5-epoch-claude-opus-4-5-20251101-32k","Claude Opus 4.5 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.2",86.04798,86.04798,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-24","2025-11-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-24","2026-09-01","2026-08-17","source-checked","choice:0.86±0.02; stderr=0.020686959671393914."],["epoch-gpqa-mEkZaYpiurQ6TEMAALK3su","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (no thinking)","claude-opus-4-5-epoch-claude-opus-4-5-20251101","Claude Opus 4.5 (no thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.2",80.681818,80.681818,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-24","2025-11-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-24","2026-09-01","2026-08-17","source-checked","choice:0.81±0.02; stderr=0.023909479653766753."],["epoch-gpqa-Cg54udQGcTCnfDGX2qePLZ","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview","gemini-3-pro-epoch-gemini-3-pro-preview","Gemini 3 Pro Preview","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.2",92.613636,92.613636,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-11-19","2025-11-19","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-11-19","2026-09-01","2026-08-18","source-checked","choice:0.93±0.02; stderr=0.016526963574741."],["epoch-gpqa-CDdjbqiYYPtEBGoZLMPM3n","deepseek-v3-2","DeepSeek V3.2","DeepSeek-V3.2 (Thinking)","deepseek-v3-2-epoch-deepseek-reasoner","DeepSeek-V3.2 (Thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",83.423521,83.423521,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-12-16","2025-12-16","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-16","2026-09-01","2026-08-17","source-checked","choice:0.83±0.02; stderr=0.0201537212608744."],["epoch-gpqa-oAu9CZSNPZgwiwKnoNpM8Q","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview","gemini-3-flash-epoch-gemini-3-flash-preview","Gemini 3 Flash Preview","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",83.207071,83.207071,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-12-17","2025-12-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-17","2026-09-01","2026-08-17","source-checked","choice:0.83±0.02; stderr=0.019036353018587762."],["epoch-gpqa-3Q6osT8EYezayG7yfxkSpr","gpt-5-2","GPT-5.2","GPT-5.2 (high)","gpt-5-2-livebench-2026-06-25-high","GPT-5.2 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",88.194444,88.194444,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-12-11","2025-12-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-11","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.01913858277633924."],["epoch-gpqa-oD4Jpoi75KWChqTPDaN9aB","gpt-5-2","GPT-5.2","GPT-5.2 (low)","gpt-5-2-epoch-gpt-5-2-2025-12-11-low","GPT-5.2 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",82.70202,82.70202,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-12-11","2025-12-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-11","2026-09-01","2026-08-17","source-checked","choice:0.83±0.02; stderr=0.02265112408975899."],["epoch-gpqa-MuhtxxvALeZZ37cSG5CVcc","gpt-5-2","GPT-5.2","GPT-5.2 (medium)","gpt-5-2-medium","GPT-5.2 (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",87.878788,87.878788,"percent","higher","p1d-research-unadmitted","excluded","direct","2025-12-11","2025-12-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-11","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.019090689343231743."],["epoch-gpqa-Ejo6Fy7XYxUDcCwbejfcYX","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",91.4,91.4,"percent","higher","2.2.0","ranking-eligible","direct","2025-12-13","2025-12-13","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-13","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.018."],["epoch-gpqa-LC5Rhog6wt2PT84CtJ483h","gpt-oss-120b","GPT-OSS 120B","gpt-oss-120b (high)","gpt-oss-120b-high","gpt-oss-120b (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.3",75.757576,75.757576,"percent","higher","2.2.0","ranking-eligible","direct","2025-12-11","2025-12-11","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2025-12-11","2026-09-01","2026-08-18","source-checked","choice:0.76±0.03; stderr=0.027437163443883767."],["epoch-gpqa-DiXCMhdfhcjaCUSzYWgACy","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (32k thinking)","claude-opus-4-6-epoch-claude-opus-4-6-32k","Claude Opus 4.6 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.4",90.530303,90.530303,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-06","2026-02-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-06","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.016916245311450143."],["epoch-gpqa-eB9e2mtd5nRwqdFyLEsz6k","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (64k thinking)","claude-opus-4-6-epoch-claude-opus-4-6-64k","Claude Opus 4.6 (64k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.4",88.762626,88.762626,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-06","2026-02-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-06","2026-09-01","2026-08-17","source-checked","choice:0.89±0.02; stderr=0.019363218328924313."],["epoch-gpqa-6phXzd4aCbnBUSQmBsaywa","glm-4-7","GLM-4.7","GLM-4.7","glm-4-7-epoch-glm-4-7","GLM-4.7","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.4",83.333333,83.333333,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-01-29","2026-01-29","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-01-29","2026-09-01","2026-08-17","source-checked","choice:0.83±0.02; stderr=0.02374901666263655."],["epoch-gpqa-FmqGUq87N4JcuSbo6Uu3U5","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Fireworks)","kimi-k2-5-epoch-fireworks-kimi-k2p5","Kimi K2.5 (Fireworks)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.4",87.6,87.6,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-02","2026-02-02","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-02","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.019."],["epoch-gpqa-kR9F3oHBCnwQW4AiTPKwpd","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview","gemini-3-1-pro-preview-epoch-gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.5",94.1,94.1,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-20","2026-02-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-20","2026-09-01","2026-08-17","source-checked","choice:0.94±0.02; stderr=0.017."],["epoch-gpqa-n4hyz75SVpwPEisKVVjvNg","glm-5","GLM-5","GLM-5","glm-5-epoch-glm-5","GLM-5","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.5",87.817259,87.817259,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-12","2026-02-12","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-12","2026-09-01","2026-08-17","source-checked","choice:0.88±0.02; stderr=0.02336331210092387."],["epoch-gpqa-58eQmyCfPa3FufhXJFvAL2","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (32k thinking)","claude-sonnet-4-6-epoch-claude-sonnet-4-6-32k","Claude Sonnet 4.6 (32k thinking)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.6",87.373737,87.373737,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-02-20","2026-02-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-02-20","2026-09-01","2026-08-17","source-checked","choice:0.87±0.02; stderr=0.020034282226331888."],["epoch-gpqa-fFatyce8UvpN7ZivdmrhAy","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.6",93.3,93.3,"percent","higher","2.2.0","ranking-eligible","direct","2026-03-06","2026-03-06","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-03-06","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.018."],["epoch-gpqa-gpt-5-4-pro-gpqa-cloud-run","gpt-5-4","GPT-5.4","GPT-5.4 Pro (xhigh)","gpt-5-4-pro-xhigh","GPT-5.4 Pro (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.6",94.6,94.6,"percent","higher","2.2.0","reference-only","direct","2026-03-20","2026-03-20","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-03-20","2026-09-01","2026-08-18","source-checked","choice:0.95±0.02; stderr=0.016."],["epoch-gpqa-AQF46S9hBhosKPEMy3fyB9","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (high)","gpt-5-4-mini-epoch-gpt-5-4-mini-2026-03-17-high","GPT-5.4 mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.6",83.585859,83.585859,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-04-15","2026-04-15","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-04-15","2026-09-01","2026-08-17","source-checked","choice:0.84±0.02; stderr=0.022274469189069883."],["epoch-gpqa-fZB3qGSohAVYzN97mniuUp","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (high)","gpt-5-4-nano-epoch-gpt-5-4-nano-2026-03-17-high","GPT-5.4 nano (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.6",78.47,78.47,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-04-14","2026-04-14","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-04-14","2026-09-01","2026-08-17","source-checked","choice:0.78±0.02; stderr=0.0244."],["epoch-gpqa-VJezHrJ5YPCbpui8pZgcJg","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (xhigh)","claude-opus-4-7-livebench-2026-06-25-xhigh","Claude Opus 4.7 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.7",90.151515,90.151515,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-04-17","2026-04-17","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-04-17","2026-09-01","2026-08-17","source-checked","choice:0.9±0.02; stderr=0.018464275887293338."],["epoch-gpqa-2CdomqtBVz2it7a2Zx2G2t","gpt-5-5","GPT-5.5","GPT-5.5 (low)","gpt-5-5-low","GPT-5.5 (low)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.8",90.656566,90.656566,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-05-05","2026-05-05","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-05-05","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.017532658377721708."],["epoch-gpqa-FUv4EUwR78nAsdGbiwiLEr","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.8",94.002525,94.002525,"percent","higher","2.2.0","ranking-eligible","direct","2026-04-24","2026-04-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-04-24","2026-09-01","2026-08-17","source-checked","choice:0.94±0.02; stderr=0.015469771177277509."],["epoch-gpqa-cBmhJzrC3SkfZ7uVgGVYZX","gpt-5-5","GPT-5.5","GPT-5.5 Pro (xhigh)","gpt-5-5-pro","GPT-5.5 Pro (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.8",93.921356,93.921356,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-04-24","2026-04-24","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-04-24","2026-09-01","2026-08-17","source-checked","choice:0.94±0.02; stderr=0.015832823810775415."],["epoch-gpqa-LjgJoDosTLkiGVT59smA7q","kimi-k2-6","Kimi K2.6","Kimi K2.6","kimi-k2-6-epoch-kimi-k2-6","Kimi K2.6","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.8",90.782828,90.782828,"percent","higher","p1d-research-unadmitted","excluded","direct","2026-05-01","2026-05-01","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-05-01","2026-09-01","2026-08-17","source-checked","choice:0.91±0.02; stderr=0.0171745162866666."],["epoch-gpqa-dYYtcwENFRtm9jZTDA2p9C","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","1.0.9",92.80303,92.80303,"percent","higher","2.2.0","ranking-eligible","direct","2026-05-22","2026-05-22","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-05-22","2026-09-01","2026-08-17","source-checked","choice:0.93±0.02; stderr=0.016387002835275905."],["evidence-2026-08-muse-glimmer-30b-gpqa-diamond-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","198-question Diamond set",83.5,83.5,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["evidence-2026-08-muse-glimmer-30b-gpqa-diamond-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","198-question Diamond set",83.5,83.5,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Meta's methodology identifies this exact High-reasoning result as sourced from Artificial Analysis."],["benchlm-ref-claude-opus-4-6-gpqadiamond-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.2,91.0116,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gpqadiamond-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.2,91.0116,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-gpqadiamond-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.2,91.0116,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqadiamond-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.2,98.1452,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqadiamond-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.2,98.1452,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-gpqadiamond-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.2,98.1452,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqadiamond-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqadiamond-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-gpqadiamond-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-gpqadiamond-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.4,88.4434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71.2,65.3303,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71.2,65.3303,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71.2,65.3303,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.1,89.4421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-gpqadiamond-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.1,90.8689,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.9,67.7557,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.9,67.7557,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.9,67.7557,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.1,92.2956,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.3,98.2879,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.3,98.2879,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.3,98.2879,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.676,95.9709,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.676,95.9709,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.676,95.9709,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqadiamond-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",78.8,76.1735,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqadiamond-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",78.8,76.1735,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-gpqadiamond-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",78.8,76.1735,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqadiamond-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86,86.446,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqadiamond-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86,86.446,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-gpqadiamond-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86,86.446,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gpqadiamond-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86.2,86.7313,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gpqadiamond-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86.2,86.7313,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-gpqadiamond-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",86.2,86.7313,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqadiamond-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",91.2,93.865,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqadiamond-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",91.2,93.865,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-gpqadiamond-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",91.2,93.865,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqadiamond-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.8,96.1478,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqadiamond-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.8,96.1478,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-gpqadiamond-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.8,96.1478,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqadiamond-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqadiamond-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-gpqadiamond-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.6,97.2892,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.3,95.4344,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.3,95.4344,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.3,95.4344,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.6,98.7159,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.6,98.7159,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",94.6,98.7159,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.9,96.2905,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.9,96.2905,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.9,96.2905,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gpqadiamond-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.5,90.0128,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gpqadiamond-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.5,90.0128,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-gpqadiamond-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",88.5,90.0128,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqadiamond-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.2,88.1581,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqadiamond-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.2,88.1581,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-gpqadiamond-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.2,88.1581,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqadiamond-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.9,89.1568,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqadiamond-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.9,89.1568,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-gpqadiamond-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.9,89.1568,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-gpqadiamond-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.5,91.4396,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqadiamond-2026-07-21","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.9,92.0103,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqadiamond-2026-07-27","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.9,92.0103,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-interfaze-beta-gpqadiamond-2026-08-01","interfaze-beta","Interfaze Beta","Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.9,92.0103,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqadiamond-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.6,88.7288,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqadiamond-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.6,88.7288,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-gpqadiamond-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87.6,88.7288,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqadiamond-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.5,92.8663,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqadiamond-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.5,92.8663,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-gpqadiamond-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.5,92.8663,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqadiamond-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.5,97.1465,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqadiamond-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.5,97.1465,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-gpqadiamond-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",93.5,97.1465,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqadiamond-2026-07-21","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",25.41,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqadiamond-2026-07-27","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",25.41,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-230m-gpqadiamond-2026-08-01","lfm2-5-230m","LFM2.5-230M","Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",25.41,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqadiamond-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",84.2,83.8779,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqadiamond-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",84.2,83.8779,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-gpqadiamond-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",84.2,83.8779,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-07-21","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",40.9,22.1002,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-07-27","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",40.9,22.1002,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-08-01","mellum2-12b-a2-5b-instruct","Mellum2-12B-A2.5B-Instruct","Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",40.9,22.1002,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-07-21","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.6,45.9267,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-07-27","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.6,45.9267,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-08-01","mellum2-12b-a2-5b-thinking","Mellum2-12B-A2.5B-Thinking","Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.6,45.9267,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-gpqadiamond-2026-07-21","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",26.26,1.2127,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-gpqadiamond-2026-07-27","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",26.26,1.2127,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minicpm5-1b-gpqadiamond-2026-08-01","minicpm5-1b","MiniCPM5-1B","Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",26.26,1.2127,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gpqadiamond-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gpqadiamond-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-gpqadiamond-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gpqadiamond-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.5,91.4396,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gpqadiamond-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.5,91.4396,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-gpqadiamond-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",89.5,91.4396,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.2,66.757,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.2,66.757,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",72.2,66.757,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",87,87.8727,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqadiamond-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.4,95.5771,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqadiamond-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.4,95.5771,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-gpqadiamond-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",92.4,95.5771,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqadiamond-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.3,92.581,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqadiamond-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.3,92.581,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-gpqadiamond-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",90.3,92.581,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqadiamond-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqadiamond-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-gpqadiamond-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",95.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-07-21","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",43.4,25.667,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-07-27","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",43.4,25.667,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-08-01","soofi-s-30b-a3b","Soofi S 30B-A3B","Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",43.4,25.667,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gpqadiamond-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",63.3,54.0591,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gpqadiamond-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",63.3,54.0591,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-gpqadiamond-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",63.3,54.0591,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gpqadiamond-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",76.3,72.6066,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gpqadiamond-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",76.3,72.6066,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-gpqadiamond-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",76.3,72.6066,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-07-21","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.3,45.4986,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-07-27","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.3,45.4986,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-08-01","zaya1-74b-preview","ZAYA1-74B-Preview","Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",57.3,45.4986,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqadiamond-2026-07-21","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71,65.0449,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqadiamond-2026-07-27","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71,65.0449,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-zaya1-8b-gpqadiamond-2026-08-01","zaya1-8b","ZAYA1-8B","Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2023",71,65.0449,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aagpqadiamond-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",37.4,23.1707,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aagpqadiamond-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",37.4,19.4602,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aagpqadiamond-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",37.4,19.4602,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aagpqadiamond-2026-07-21","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.9,38.7534,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aagpqadiamond-2026-07-27","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.9,35.7955,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aagpqadiamond-2026-08-01","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.9,35.7955,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,65.0407,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,63.3523,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,63.3523,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,82.1138,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aagpqadiamond-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.6,97.9675,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aagpqadiamond-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.6,97.8693,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aagpqadiamond-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.6,97.8693,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81,82.2493,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81,81.392,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81,81.392,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.8374,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.3466,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.3466,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aagpqadiamond-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.9024,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aagpqadiamond-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.608,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aagpqadiamond-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.608,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.5,92.4119,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.5,92.0455,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.5,92.0455,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aagpqadiamond-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",93.2,98.7216,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aagpqadiamond-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",93.2,98.7216,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.9,80.7588,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.9,79.8295,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.9,79.8295,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.1,95.935,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.1,95.7386,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.1,95.7386,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aagpqadiamond-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.1,75.6098,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aagpqadiamond-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.1,74.4318,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aagpqadiamond-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.1,74.4318,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-07-21","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",61.5,55.8266,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-07-27","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",61.5,53.6932,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-08-01","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",61.5,53.6932,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aagpqadiamond-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",55.7,47.9675,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aagpqadiamond-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",55.7,45.4545,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aagpqadiamond-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",55.7,45.4545,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aagpqadiamond-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",77.9,78.0488,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aagpqadiamond-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",77.9,76.9886,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aagpqadiamond-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",77.9,76.9886,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.5,72.0867,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.5,70.7386,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.5,70.7386,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.1,74.2547,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.1,73.0114,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.1,73.0114,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aagpqadiamond-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.3,82.6558,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aagpqadiamond-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.3,81.8182,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aagpqadiamond-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.3,81.8182,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.4,29.9458,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.4,26.5625,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.4,26.5625,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",62.8,57.5881,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",62.8,55.5398,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",62.8,55.5398,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-07-21","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",27.7,10.0271,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-07-27","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",27.7,5.6818,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-08-01","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",27.7,5.6818,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-07-21","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.9,52.3035,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-07-27","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.9,50,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-08-01","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.9,50,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,65.0407,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,63.3523,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.3,63.3523,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.4,86.8564,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.4,86.2216,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.4,86.2216,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aagpqadiamond-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.2,82.5203,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aagpqadiamond-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.2,81.6761,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aagpqadiamond-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.2,81.6761,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aagpqadiamond-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.8,95.5285,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aagpqadiamond-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.8,95.3125,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aagpqadiamond-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.8,95.3125,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.2,83.8753,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.2,83.0966,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.2,83.0966,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",83.8,86.0434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",83.8,85.3693,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",83.8,85.3693,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.8,98.2385,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.8,98.1534,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.8,98.1534,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aagpqadiamond-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.8,30.4878,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aagpqadiamond-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.8,27.1307,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aagpqadiamond-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.8,27.1307,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.2,79.8103,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.2,78.8352,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",79.2,78.8352,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aagpqadiamond-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.6179,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aagpqadiamond-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.0682,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aagpqadiamond-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.0682,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",37.5,23.3062,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",40.5,23.8636,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",40.5,23.8636,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.6,50.542,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",52.2,40.483,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",52.2,40.483,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aagpqadiamond-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.3,71.8157,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aagpqadiamond-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.3,70.4545,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aagpqadiamond-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.3,70.4545,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aagpqadiamond-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.2,58.1301,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aagpqadiamond-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.2,56.108,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aagpqadiamond-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.2,56.108,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aagpqadiamond-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.9,88.8889,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aagpqadiamond-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.9,88.3523,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aagpqadiamond-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.9,88.3523,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aagpqadiamond-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,87.2629,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aagpqadiamond-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,86.6477,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aagpqadiamond-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,86.6477,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,82.1138,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aagpqadiamond-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.6,62.7371,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aagpqadiamond-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.6,60.9375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aagpqadiamond-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.6,60.9375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.4,62.4661,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.4,60.6534,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",66.4,60.6534,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.2,41.8699,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.2,39.0625,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.2,39.0625,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aagpqadiamond-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",54.3,46.0705,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aagpqadiamond-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",54.3,43.4659,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aagpqadiamond-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",54.3,43.4659,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.6,30.2168,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.6,26.8466,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",42.6,26.8466,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,88.2114,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,87.642,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,87.642,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,86.5854,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,88.2114,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,87.642,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aagpqadiamond-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.4,87.642,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,86.5854,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aagpqadiamond-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aagpqadiamond-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.3,90.7859,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aagpqadiamond-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.3,90.3409,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aagpqadiamond-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.3,90.3409,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,89.0244,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,88.4943,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,88.4943,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,89.0244,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,88.4943,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86,88.4943,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aagpqadiamond-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.3,94.8509,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aagpqadiamond-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.3,94.6023,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aagpqadiamond-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.3,94.6023,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.9,94.3089,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.9,94.0341,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.9,94.0341,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.5,96.477,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.5,96.3068,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",91.5,96.3068,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.5,91.0569,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.5,90.625,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.5,90.625,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.7,83.1978,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.7,82.3864,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",81.7,82.3864,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.2,78.4553,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.2,77.4148,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.2,77.4148,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.8,65.7182,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.8,64.0625,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68.8,64.0625,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.1,10.5691,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.1,6.25,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.1,6.25,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",20.3,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",23.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",23.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",26.3,8.1301,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",24.6,1.2784,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",24.6,1.2784,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",25.7,7.3171,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",29,7.5284,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",29,7.5284,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aagpqadiamond-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.7,91.3279,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aagpqadiamond-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.7,90.9091,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aagpqadiamond-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.7,90.9091,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,87.2629,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,86.6477,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.7,86.6477,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.7,58.8076,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.7,56.8182,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.7,56.8182,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.3,88.0759,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.3,87.5,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.3,87.5,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aagpqadiamond-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.1,94.5799,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aagpqadiamond-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.1,94.3182,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aagpqadiamond-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",90.1,94.3182,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aagpqadiamond-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",93.1,98.645,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aagpqadiamond-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",93.1,98.5795,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aagpqadiamond-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",93.1,98.5795,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.7,71.0027,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.7,69.6023,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.7,69.6023,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aagpqadiamond-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.7,94.0379,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aagpqadiamond-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.7,93.75,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aagpqadiamond-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.7,93.75,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aagpqadiamond-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.3,78.5908,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aagpqadiamond-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.3,77.5568,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aagpqadiamond-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",78.3,77.5568,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aagpqadiamond-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.6,76.2873,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aagpqadiamond-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.6,75.142,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aagpqadiamond-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.6,75.142,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aagpqadiamond-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.9,91.5989,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aagpqadiamond-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.9,91.1932,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aagpqadiamond-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87.9,91.1932,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.9024,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.608,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.6,93.608,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.3,42.0054,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",46.6,32.5284,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",46.6,32.5284,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.9,11.6531,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.9,7.3864,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",28.9,7.3864,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",59.3,52.8455,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",59.3,50.5682,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",59.3,50.5682,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.5,42.2764,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.5,39.4886,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",51.5,39.4886,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aagpqadiamond-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",67.1,63.4146,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aagpqadiamond-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",67.1,61.6477,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aagpqadiamond-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",67.1,61.6477,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aagpqadiamond-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.7,52.0325,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aagpqadiamond-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.7,49.7159,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aagpqadiamond-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",58.7,49.7159,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",65.6,61.3821,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",65.6,59.517,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",65.6,59.517,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.8,84.6883,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.8,83.9489,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.8,83.9489,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87,90.3794,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87,89.9148,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",87,89.9148,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.8374,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.3466,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",86.6,89.3466,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aagpqadiamond-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.9,98.374,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aagpqadiamond-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.9,98.2955,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aagpqadiamond-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",92.9,98.2955,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aagpqadiamond-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.6,38.3469,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aagpqadiamond-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.6,35.3693,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aagpqadiamond-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",48.6,35.3693,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aagpqadiamond-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68,64.6341,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aagpqadiamond-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68,62.9261,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aagpqadiamond-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",68,62.9261,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aagpqadiamond-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.8,50.813,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aagpqadiamond-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.8,48.4375,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aagpqadiamond-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.8,48.4375,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,73.8482,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,72.5852,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,72.5852,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aagpqadiamond-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,76.6938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aagpqadiamond-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,75.5682,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aagpqadiamond-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,75.5682,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aagpqadiamond-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,76.6938,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aagpqadiamond-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,75.5682,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aagpqadiamond-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.9,75.5682,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.8,94.1734,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.8,93.892,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.8,93.892,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.7,75.0678,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.7,73.8636,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",75.7,73.8636,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.8,71.1382,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.8,69.7443,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",72.8,69.7443,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aagpqadiamond-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",49.9,40.1084,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aagpqadiamond-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",49.9,37.2159,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aagpqadiamond-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",49.9,37.2159,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aagpqadiamond-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.7,73.7127,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aagpqadiamond-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.7,72.4432,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aagpqadiamond-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.7,72.4432,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aagpqadiamond-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.7,84.5528,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aagpqadiamond-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.7,83.8068,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aagpqadiamond-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",82.7,83.8068,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aagpqadiamond-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,73.8482,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aagpqadiamond-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,72.5852,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aagpqadiamond-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",74.8,72.5852,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-pro-aagpqadiamond-2026-07-21","o3-pro","o3-pro","Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.9919,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-pro-aagpqadiamond-2026-07-27","o3-pro","o3-pro","Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.3636,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-pro-aagpqadiamond-2026-08-01","o3-pro","o3-pro","Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.3636,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aagpqadiamond-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.5,50.4065,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aagpqadiamond-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.5,48.0114,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aagpqadiamond-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",57.5,48.0114,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.8,92.8184,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.8,92.4716,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.8,92.4716,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-07-21","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",41.7,28.9973,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-07-27","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",41.7,25.5682,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-08-01","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",41.7,25.5682,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aagpqadiamond-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.4,76.0163,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aagpqadiamond-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.4,74.858,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aagpqadiamond-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",76.4,74.858,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.8,88.7534,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.8,88.2102,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.8,88.2102,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aagpqadiamond-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.4959,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aagpqadiamond-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.1818,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aagpqadiamond-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.1818,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.4959,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.1818,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",89.3,93.1818,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.6179,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.0682,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",85.7,88.0682,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.9919,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.3636,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.5,86.3636,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,86.5854,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.2,85.9375,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.2,92.0054,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.2,91.6193,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",88.2,91.6193,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.1,86.4499,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.1,85.7955,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",84.1,85.7955,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aagpqadiamond-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.8,72.4932,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aagpqadiamond-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.8,71.1648,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aagpqadiamond-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",73.8,71.1648,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aagpqadiamond-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.3,58.2656,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aagpqadiamond-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.3,56.25,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aagpqadiamond-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",63.3,56.25,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aagpqadiamond-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",56.1,48.5095,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aagpqadiamond-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",56.1,46.0227,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aagpqadiamond-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",56.1,46.0227,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aagpqadiamond-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,82.1138,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aagpqadiamond-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aagpqadiamond-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","2026",80.9,81.25,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["aa-individual:claude-fable-5:gpqa-diamond:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.6262626262626,92.6262626262626,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:claude-opus-4-6-thinking:gpqa-diamond:2026-08-29","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",89.59595959596,89.59595959596,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-6-thinking-2026-08-29","aa-current-claude-opus-4-6-thinking-2026-08-29","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-6-adaptive","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:claude-opus-4-7-adaptive:gpqa-diamond:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.414141414141,91.414141414141,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:claude-opus-4-8:gpqa-diamond:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.020202020202,92.020202020202,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:gpqa-diamond:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",93.2323232323232,93.2323232323232,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-sonnet-5:gpqa-diamond:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.1111111111111,91.1111111111111,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:deepseek-v3-1:gpqa-diamond:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","deepseek-v3-1-non-reasoning","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",73.535353535353,73.535353535353,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-2026-08-29","aa-current-deepseek-v3-1-2026-08-29","DeepSeek V3.1 (Non-reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:deepseek-v3-1-reasoning:gpqa-diamond:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","deepseek-v3-1-reasoning-default","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",77.878787878788,77.878787878788,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-reasoning-2026-08-29","aa-current-deepseek-v3-1-reasoning-2026-08-29","DeepSeek V3.1 (Reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:deepseek-v4-flash-vision-exp:gpqa-diamond:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.3131313131313,91.3131313131313,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual gpqa-diamond result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-individual:deepseek-v4-pro-0813:gpqa-diamond:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.82828282828281,92.82828282828281,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-6-flash:gpqa-diamond:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.82828282828281,92.82828282828281,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:gpqa-diamond:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.12121212121211,92.12121212121211,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:gpqa-diamond:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",42.828282828283,42.828282828283,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gemma-4-26b-a4b:gpqa-diamond:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",79.191919191919,79.191919191919,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:glm-5-2:gpqa-diamond:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",89.49494949495,89.49494949495,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:gpqa-diamond:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.7171717171717,91.7171717171717,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:gpqa-diamond:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.2121212121212,91.2121212121212,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual gpqa-diamond result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:gpqa-diamond:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",66.363636363636,66.363636363636,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:gpqa-diamond:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",51.212121212121,51.212121212121,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:gpt-5-4:gpqa-diamond:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.020202020202,92.020202020202,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-5:gpqa-diamond:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",93.53535353535351,93.53535353535351,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-luna:gpqa-diamond:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",91.1111111111111,91.1111111111111,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-sol:gpqa-diamond:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",94.1414141414141,94.1414141414141,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-terra:gpqa-diamond:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.5252525252525,92.5252525252525,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-5:gpqa-diamond:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",93.13131313131309,93.13131313131309,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-6:gpqa-diamond:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",94.94949494949499,94.94949494949499,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:kimi-k2-5-reasoning:gpqa-diamond:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",87.878787878788,87.878787878788,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["new-model:kimi-k2-5-model-card-2026-08-29:kimi-k2-5-reasoning:gpqa-diamond:cell:kimi-k2-5:gpqa-diamond:thinking","kimi-k2-5","Kimi K2.5","Kimi K2.5 thinking configuration","kimi-k2-5-thinking","Kimi K2.5 thinking configuration","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",87.6,87.6,"percent","higher","2.1.0","ranking-eligible","direct","2026-01-27",null,"production::kimi-k2-5-model-card-2026-08-29","kimi-k2-5-model-card-2026-08-29","Kimi K2.5 official model card and evaluation table","Moonshot AI","https://huggingface.co/moonshotai/Kimi-K2.5","2026-01-27","2026-08-29","2026-08-29","source-checked","No tools; 96K maximum generation; avg@8."],["aa-individual:kimi-k3:gpqa-diamond:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",93.53535353535351,93.53535353535351,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:llama-4-maverick:gpqa-diamond:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",67.070707070707,67.070707070707,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:gpqa-diamond:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",58.686868686869,58.686868686869,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:gpqa-diamond:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",77.979797979798,77.979797979798,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-current:mistral-medium-3-5-128b:gpqa-diamond:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",74.848484848485,74.848484848485,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:mistral-small-4-reasoning:gpqa-diamond:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",76.868686868687,76.868686868687,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:muse-spark-1-1:gpqa-diamond:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",89.79797979797979,89.79797979797979,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:muse-spark-1-2:gpqa-diamond:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",90.4040404040404,90.4040404040404,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual gpqa-diamond result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:gpqa-diamond:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",75.656565656566,75.656565656566,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-397b-reasoning:gpqa-diamond:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",89.292929292929,89.292929292929,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:gpqa-diamond:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",85.656565656566,85.656565656566,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-6-35b-a3b:gpqa-diamond:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",84.141414141414,84.141414141414,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen-3-8-flash-next:gpqa-diamond:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",92.323232323232,92.323232323232,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:gpqa-diamond:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","diamond",75.151515151515,75.151515151515,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:gpqa:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",91.3,91.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-gpqa-diamond-diamond-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",91.3,91.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-command-a-plus-gpqa-diamond-diamond-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",75.6,75.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-gpqa-diamond-diamond-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",88.9,88.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:gpqa:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",90.8,90.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-longcat-2-0-gpqa-diamond-diamond","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",88.9,88.9,"percent","higher","2.1.0","ranking-eligible","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score."],["evidence-2026-08-15-mimo-v2-5-gpqa-diamond-diamond-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",83,83,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-gpqa-diamond-diamond-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",77.5,77.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-qwen3-6-27b-gpqa-diamond-diamond-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",87.8,87.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:gpqa:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",90.3,90.3,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-gpqa-diamond-diamond-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",90.3,90.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:gpqa:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",89.2,89.2,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-gpqa-diamond-diamond","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",89.2,89.2,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["new-model:aa-parity-model-qwen-3-8-flash-next-2026-08-27:qwen-3-8-flash-next:gpqa-diamond:cell:language:gpqa:0","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (xhigh default thinking configuration)","qwen-3-8-flash-next-xhigh","Qwen3.8-Flash-Next (xhigh default thinking configuration)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",91.7,91.7,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-26",null,"production::qwen3-8-flash-next-release-current-2026-08-27","qwen3-8-flash-next-release-current-2026-08-27","Qwen3.8-Flash-Next current official launch page","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-27","2026-08-29","source-checked","Generic v2 table-provenance repair: the row retains the exact official Qwen table source and event supplied by its comparison table."],["evidence-2026-08-15-solar-open-100b-reasoning-gpqa-diamond-diamond-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",66.2,66.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-gpqa-diamond-diamond","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors","Diamond",86.3,86.3,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-057","claude-fable-5","Claude Fable 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.63,92.63,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-fable-5-evals","aa-claude-fable-5-evals","Artificial Analysis evaluations for claude-fable-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-067","claude-haiku-4-5","Claude Haiku 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,64.65,64.65,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-claude-4-5-haiku-evals","aa-claude-4-5-haiku-evals","Artificial Analysis evaluations for claude-4-5-haiku","Artificial Analysis","https://artificialanalysis.ai/models/claude-4-5-haiku","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-135","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,94.1,94.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1404","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,79.596,79.596,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1393","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,80.9,80.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-609","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,87,87,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-605","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.3,91.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-060","claude-opus-4-8","Claude Opus 4.8","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.02,92.02,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-opus-4-8-evals","aa-claude-opus-4-8-evals","Artificial Analysis evaluations for claude-opus-4-8","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-136","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.6,93.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1776","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,77.1717,77.1717,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1726","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,77.6768,77.6768,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1380","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.4343,83.4343,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-608","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.9,89.9,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-063","claude-sonnet-5","Claude Sonnet 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.11,91.11,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-sonnet-5-evals","aa-claude-sonnet-5-evals","Artificial Analysis evaluations for claude-sonnet-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1634","command-a-plus","Command A+","Command A+",null,"Command A+","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,76.0606,76.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1540","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,79.1919,79.1919,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1526","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.0404,84.0404,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-082","deepseek-v4-flash","DeepSeek V4 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.39,89.39,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-flash-evals","aa-deepseek-v4-flash-evals","Artificial Analysis evaluations for deepseek-v4-flash","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-147","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,71.2,71.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-079","deepseek-v4-pro","DeepSeek V4 Pro","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,88.79,88.79,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-pro-evals","aa-deepseek-v4-pro-evals","Artificial Analysis evaluations for deepseek-v4-pro","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-146","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,72.9,72.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1838","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,79.2929,79.2929,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1800","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.4444,84.4444,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-071","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,94.14,94.14,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gemini-3-1-pro-preview-evals","aa-gemini-3-1-pro-preview-evals","Artificial Analysis evaluations for gemini-3-1-pro-preview","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-1-pro-preview","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-116","gemini-3-5-flash","Gemini 3.5 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.12,92.12,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gemini-3-5-flash-medium-evals","aa-gemini-3-5-flash-medium-evals","Artificial Analysis evaluations for gemini-3-5-flash-medium","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-medium","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-142","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.2,92.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2039","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis GPQA Diamond independent evaluation.",null,"Artificial Analysis GPQA Diamond independent evaluation.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.8,83.8,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA GPQA Diamond 83.8%."],["evidence-2026-07-1999","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis GPQA Diamond independent evaluation.",null,"Artificial Analysis GPQA Diamond independent evaluation.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.8,92.8,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA GPQA Diamond 92.8%."],["evidence-2026-07-1882","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,75.2525,75.2525,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1621","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,79.1919,79.1919,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-611","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.3,84.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1741","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,77.9798,77.9798,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1554","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,85.8586,85.8586,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-610","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,86,86,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-606","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.2,91.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1340","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,85.3535,85.3535,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1340--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,85.3535,85.3535,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1325","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.7374,83.7374,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1325--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.7374,83.7374,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1352","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,80.303,80.303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1352--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,80.303,80.303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1304","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.303,90.303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1304--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.303,90.303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1315","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.899,89.899,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1315--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.899,89.899,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-105","gpt-5-3-codex","GPT-5.3-Codex","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.52,91.52,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-3-codex-evals","aa-gpt-5-3-codex-evals","Artificial Analysis evaluations for gpt-5-3-codex","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-3-codex","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-101","gpt-5-4","GPT-5.4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.02,92.02,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-4-evals","aa-gpt-5-4-evals","Artificial Analysis evaluations for gpt-5-4","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-139","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.8,92.8,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1294","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,87.4747,87.4747,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1294--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,87.4747,87.4747,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-612","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,82.8,82.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-112","gpt-5-5","GPT-5.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.54,93.54,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-5-evals","aa-gpt-5-5-evals","Artificial Analysis evaluations for gpt-5-5","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-137","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.6,93.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-053","gpt-5-6-luna","GPT-5.6 Luna","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.11,91.11,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-luna-evals","aa-gpt-5-6-luna-evals","Artificial Analysis evaluations for gpt-5-6-luna","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-141","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.3,92.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-045","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,94.14,94.14,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-sol-evals","aa-gpt-5-6-sol-evals","Artificial Analysis evaluations for gpt-5-6-sol","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-134","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,94.6,94.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-049","gpt-5-6-terra","GPT-5.6 Terra","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.53,92.53,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-terra-evals","aa-gpt-5-6-terra-evals","Artificial Analysis evaluations for gpt-5-6-terra","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-138","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.9,92.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1871","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,79.0909,79.0909,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1659","grok-4","Grok 4","Grok 4",null,"Grok 4","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,87.6768,87.6768,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1763","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.7475,84.7475,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1448","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.1111,91.1111,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-075","grok-4-3","Grok 4.3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.1,90.1,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis evaluations for grok-4-3","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-144","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.1,90.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-119","grok-4-5","Grok 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.13,93.13,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-5-evals","aa-grok-4-5-evals","Artificial Analysis evaluations for grok-4-5","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1893","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,72.7273,72.7273,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1900","hy3","Hy3","tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.",null,"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.4,90.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-06","2026-07-06","production::tencent-hy3-hf","tencent-hy3-hf","Tencent Hy3 model card on Hugging Face","Tencent Hy Team","https://huggingface.co/tencent/Hy3","2026-07-06","2026-07-16","2026-07-16","provider-reported","GPQA Diamond score listed in Hugging Face evaluation results for tencent/Hy3."],["evidence-2026-07-1850","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,76.6667,76.6667,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1671","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.8384,83.8384,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-145","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,87.6,87.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1422","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,91.1111,91.1111,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1438","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.596,89.596,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1942","kimi-k3","Kimi K3","GPQA Diamond; max reasoning; provider-published Kimi K3 evaluation.",null,"GPQA Diamond; max reasoning; provider-published Kimi K3 evaluation.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.5,93.5,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 93.5% on GPQA Diamond."],["evidence-2026-07-1942--configuration--kimi-k3-max","kimi-k3","Kimi K3","GPQA Diamond; max reasoning; provider-published Kimi K3 evaluation.","kimi-k3-max","GPQA Diamond; max reasoning; provider-published Kimi K3 evaluation.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,93.5,93.5,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 93.5% on GPQA Diamond."],["evidence-2026-07-1788","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,75.1515,75.1515,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1590","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.6465,84.6465,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1577","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.9495,84.9495,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-1752","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,77.6768,77.6768,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1691","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,83.0303,83.0303,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1565","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.8485,84.8485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-108","minimax-m3","MiniMax M3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.93,92.93,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-minimax-m3-evals","aa-minimax-m3-evals","Artificial Analysis evaluations for minimax-m3","Artificial Analysis","https://artificialanalysis.ai/models/minimax-m3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-093","mistral-large-3","Mistral Large 3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,67.98,67.98,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-large-3-evals","aa-mistral-large-3-evals","Artificial Analysis evaluations for mistral-large-3","Artificial Analysis","https://artificialanalysis.ai/models/mistral-large-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-085","mistral-medium-3-5","Mistral Medium 3.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,74.85,74.85,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-medium-3-5-evals","aa-mistral-medium-3-5-evals","Artificial Analysis evaluations for mistral-medium-3-5","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-089","mistral-small-4","Mistral Small 4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,76.87,76.87,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-small-4-evals","aa-mistral-small-4-evals","Artificial Analysis evaluations for mistral-small-4","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1916","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.8,89.8,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1916--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.8,89.8,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1829","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,80,80,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1607","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,86.6667,86.6667,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1647","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,78.4848,78.4848,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-gpqa-diamond-no-tools-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,75.44,75.44,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1861","o1","o1","o1",null,"o1","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,74.7475,74.7475,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1365","o3","o3","o3",null,"o3","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,82.7273,82.7273,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1357","o3-pro","o3-pro","o3-pro",null,"o3-pro","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.5455,84.5455,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3-pro."],["evidence-2026-07-1812","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,78.3838,78.3838,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1812--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,78.3838,78.3838,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1682","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,86.0606,86.0606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1501","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,85.6566,85.6566,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1512","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,85.7576,85.7576,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1713","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.5455,84.5455,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1487","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,89.2929,89.2929,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1702","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,82.6263,82.6263,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1461","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,84.2424,84.2424,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1471","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,88.7879,88.7879,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-607","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.4,90.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-098","qwen-3-7-max","Qwen3.7-Max","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.32,92.32,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-qwen3-7-max-evals","aa-qwen3-7-max-evals","Artificial Analysis evaluations for qwen3-7-max","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-7-max","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-140","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.4,92.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-143","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,90.3,90.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-qwen-3-8-max-gpqa-diamond","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.6,92.6,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 92.6 on GPQA Diamond."],["evidence-2026-08-qwen-3-8-max-gpqa-diamond--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","gpqa-diamond","GPQA Diamond","reasoning","GPQA authors",null,92.6,92.6,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 92.6 on GPQA Diamond."],["benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-07-21","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-graphwalksbfs128k","Graphwalks BFS 0K-128K","reasoning","OpenAI","2026",90,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-07-27","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-graphwalksbfs128k","Graphwalks BFS 0K-128K","reasoning","OpenAI","2026",90,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-08-01","mai-thinking-1","MAI-Thinking-1","Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","benchlm-graphwalksbfs128k","Graphwalks BFS 0K-128K","reasoning","OpenAI","2026",90,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",85.7,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",85.7,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",85.7,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",88,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",88,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-hellaswag","HellaSwag","reasoning","DeepSeek-AI","2026",88,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-hle-verified-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","hle-verified","HLE-Verified","reasoning","HLE-Verified authors","1,811-item verified set",31,31,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-hle-verified-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","hle-verified","HLE-Verified","reasoning","HLE-Verified authors","1,811-item verified set",51.2,51.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-hle-verified-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","hle-verified","HLE-Verified","reasoning","HLE-Verified authors","1,811-item verified set",53.6,53.6,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-hle-verified-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","hle-verified","HLE-Verified","reasoning","HLE-Verified authors","1,811-item verified set",51.1,51.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["benchlm-ref-agents-a1-hle-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",47.6,70.2465,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-hle-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",47.6,70,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-hle-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",47.6,70,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hle-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",64.5,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hle-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",64.5,99.6491,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-hle-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",64.5,99.6491,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hle-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.8,40.669,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hle-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.8,40.5263,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-hle-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.8,40.5263,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hle-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",53,79.7535,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hle-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",53,79.4737,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-hle-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",53,79.4737,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hle-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.7465,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hle-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.4561,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-hle-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.4561,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hle-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.9,88.3803,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hle-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.9,88.0702,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-hle-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.9,88.0702,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hle-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",64.7,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-hle-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",64.7,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-hle-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",49,72.7113,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-hle-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",49,72.4561,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-hle-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",49,72.4561,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hle-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.4,87.5,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hle-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.4,87.193,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-hle-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.4,87.193,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.2042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.2042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.0702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.0702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.0702,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-hle-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",29.4,38.0702,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.7113,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.5439,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.5439,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hle-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",8.1,0.7042,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hle-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",8.1,0.7018,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-hle-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",8.1,0.7018,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.7113,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.5439,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-hle-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.8,47.5439,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.1831,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.1831,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.0175,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.0175,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.0175,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-hle-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.5,47.0175,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.8169,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hle-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",7.7,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hle-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",7.7,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-hle-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",7.7,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.8169,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-hle-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-hle-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",18.8,19.5423,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-hle-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",18.8,19.4737,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-hle-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",18.8,19.4737,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-hle-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",40.2,57.2183,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-hle-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",40.2,57.0175,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-hle-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",40.2,57.0175,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hle-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",17.2,16.7254,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hle-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",17.2,16.6667,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-hle-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",17.2,16.6667,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hle-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.5,33.0986,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hle-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.5,32.9825,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-hle-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.5,32.9825,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-hle-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24.8,30.1056,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-hle-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24.8,30,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-hle-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24.8,30,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hle-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,75.1761,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hle-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,74.9123,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-hle-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,74.9123,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hle-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.3,78.5211,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hle-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.3,78.2456,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-hle-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.3,78.2456,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hle-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.7465,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hle-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.4561,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-hle-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",54.7,82.4561,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hle-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",58.7,89.7887,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hle-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",58.7,89.4737,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-hle-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",58.7,89.4737,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hle-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.1,78.169,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hle-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.1,77.8947,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-hle-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.1,77.8947,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hle-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.5,59.507,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hle-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.5,59.2982,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-hle-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.5,59.2982,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hle-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.8169,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hle-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-hle-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",37.7,52.6316,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hle-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.2,87.1479,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hle-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.2,86.8421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-hle-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",57.2,86.8421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hle-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.2,78.3451,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hle-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.2,78.0702,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-hle-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",52.2,78.0702,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-hle-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",35,48.0634,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-hle-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",35,47.8947,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-hle-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",35,47.8947,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-hle-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",25.5,31.338,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-hle-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",25.5,31.2281,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-hle-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",25.5,31.2281,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hle-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",46,67.4296,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hle-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",46,67.193,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-hle-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",46,67.193,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-hle-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",47.8,70.3509,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hle-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.1,39.4366,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hle-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.1,39.2982,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-hle-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",30.1,39.2982,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hle-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.5352,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hle-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.3684,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-hle-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.3684,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hle-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",56,85.0352,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hle-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",56,84.7368,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hle-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",48,70.9507,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hle-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",48,70.7018,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-hle-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",48,70.7018,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hle-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,75.1761,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hle-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,74.9123,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-hle-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",50.4,74.9123,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hle-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",62.1,95.7746,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hle-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",62.1,95.4386,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-hle-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",62.1,95.4386,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hle-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.7,33.4507,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hle-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.7,33.3333,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-hle-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",26.7,33.3333,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hle-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.7,36.9718,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hle-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.7,36.8421,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-hle-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.7,36.8421,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hle-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24,28.6972,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hle-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24,28.5965,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-hle-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",24,28.5965,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hle-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.8,37.1479,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hle-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.8,37.0175,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-hle-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",28.8,37.0175,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hle-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",21.4,24.1197,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hle-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",21.4,24.0351,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-hle-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",21.4,24.0351,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hle-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.4,59.331,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hle-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.4,59.1228,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-hle-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",41.4,59.1228,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hle-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.5352,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hle-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.3684,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-hle-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2025",34.7,47.3684,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aahle-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aahle-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aahle-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aahle-2026-07-21","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.1,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aahle-2026-07-27","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.1,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-opus-aahle-2026-08-01","claude-3-opus","Claude 3 Opus","Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.1,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aahle-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aahle-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aahle-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aahle-2026-07-21","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.9,17.5299,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aahle-2026-07-27","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.9,17.5299,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-1-opus-thinking-aahle-2026-08-01","claude-4-1-opus-thinking","Claude 4.1 Opus Thinking","Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.9,17.5299,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aahle-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",53.3,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aahle-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",53.3,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aahle-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",53.3,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aahle-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.4,50.3984,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aahle-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.4,50.3984,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aahle-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.4,50.3984,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aahle-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",36.7,66.9323,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aahle-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",36.7,66.9323,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aahle-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",36.7,66.9323,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aahle-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.2,55.9761,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aahle-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.2,55.9761,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aahle-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.2,55.9761,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aahle-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.4,16.5339,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aahle-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.4,16.5339,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aahle-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.4,16.5339,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-07-21","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.5,4.7809,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-07-27","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.5,4.7809,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-08-01","deepseek-r1-distill-qwen-32b","DeepSeek R1 Distill Qwen 32B","Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.5,4.7809,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aahle-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.6,0.996,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aahle-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.6,0.996,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aahle-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.6,0.996,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aahle-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13,19.7211,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aahle-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13,19.7211,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aahle-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13,19.7211,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aahle-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.3,6.3745,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aahle-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.3,6.3745,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aahle-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.3,6.3745,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aahle-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.5,14.741,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aahle-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.5,14.741,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aahle-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.5,14.741,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aahle-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.9,23.506,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aahle-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.9,23.506,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aahle-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.9,23.506,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aahle-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.8,5.3785,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aahle-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.8,5.3785,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aahle-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.8,5.3785,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aahle-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aahle-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aahle-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aahle-2026-07-21","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aahle-2026-07-27","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-0-pro-aahle-2026-08-01","gemini-1-0-pro","Gemini 1.0 Pro","Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aahle-2026-07-21","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aahle-2026-07-27","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-1-5-pro-aahle-2026-08-01","gemini-1-5-pro","Gemini 1.5 Pro","Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.9,3.5857,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aahle-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aahle-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aahle-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aahle-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.1,21.9124,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aahle-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.1,21.9124,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aahle-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.1,21.9124,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aahle-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aahle-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aahle-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aahle-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",16.2,26.0956,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aahle-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",16.2,26.0956,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aahle-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",16.2,26.0956,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aahle-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",44.7,82.8685,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aahle-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",44.7,82.8685,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aahle-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",44.7,82.8685,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aahle-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.5,28.6853,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aahle-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.5,28.6853,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aahle-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.5,28.6853,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aahle-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",38.3,70.1195,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aahle-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",38.3,70.1195,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aahle-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",38.3,70.1195,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aahle-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.7,3.1873,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aahle-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.7,3.1873,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aahle-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.7,3.1873,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aahle-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.8,23.3068,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aahle-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.8,23.3068,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aahle-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.8,23.3068,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aahle-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aahle-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aahle-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aahle-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.7,1.1952,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aahle-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.7,1.1952,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aahle-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.7,1.1952,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aahle-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.8,7.3705,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aahle-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.8,7.3705,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aahle-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.8,7.3705,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aahle-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.2,4.1833,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aahle-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.2,4.1833,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aahle-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.2,4.1833,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aahle-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",25.4,44.4223,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aahle-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",25.4,44.4223,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aahle-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",25.4,44.4223,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aahle-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",15.8,25.2988,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aahle-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",15.8,25.2988,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aahle-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",15.8,25.2988,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aahle-2026-07-21","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aahle-2026-07-27","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-turbo-aahle-2026-08-01","gpt-4-turbo","GPT-4 Turbo","Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aahle-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aahle-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aahle-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aahle-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aahle-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aahle-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.6,2.988,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aahle-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aahle-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aahle-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.9,1.5936,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aahle-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aahle-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aahle-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.3,0.3984,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aahle-2026-07-21","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aahle-2026-07-27","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-mini-aahle-2026-08-01","gpt-4o-mini","GPT-4o mini","Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aahle-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aahle-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.5,40.6375,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aahle-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aahle-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aahle-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",26.5,46.6135,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aahle-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aahle-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aahle-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aahle-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aahle-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aahle-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aahle-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",35.4,64.3426,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aahle-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",35.4,64.3426,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aahle-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",35.4,64.3426,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aahle-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",33.5,60.5578,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aahle-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",33.5,60.5578,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aahle-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",33.5,60.5578,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aahle-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",39.9,73.3068,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aahle-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",39.9,73.3068,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aahle-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",39.9,73.3068,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aahle-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aahle-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aahle-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.2,67.9283,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aahle-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",47.2,87.8486,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aahle-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",47.2,87.8486,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aahle-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",47.2,87.8486,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aahle-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",41.8,77.0916,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aahle-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",41.8,77.0916,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aahle-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",41.8,77.0916,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aahle-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",18.5,30.6773,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aahle-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",18.5,30.6773,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aahle-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",18.5,30.6773,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aahle-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.8,13.3466,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aahle-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.8,13.3466,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aahle-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.8,13.3466,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aahle-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aahle-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aahle-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aahle-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.7,5.1793,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aahle-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.7,5.1793,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aahle-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.7,5.1793,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aahle-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aahle-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aahle-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aahle-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.4,6.5737,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aahle-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.4,6.5737,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aahle-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.4,6.5737,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aahle-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.9,41.4343,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aahle-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.9,41.4343,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aahle-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.9,41.4343,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aahle-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17,27.6892,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aahle-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17,27.6892,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aahle-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17,27.6892,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aahle-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aahle-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aahle-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5,3.7849,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.6,28.8845,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.6,28.8845,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",17.6,28.8845,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aahle-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",40.3,74.1036,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aahle-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",40.3,74.1036,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aahle-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",40.3,74.1036,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aahle-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.5,8.7649,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aahle-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.5,8.7649,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aahle-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.5,8.7649,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aahle-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.6,56.7729,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aahle-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.6,56.7729,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aahle-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",31.6,56.7729,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aahle-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13.1,19.9203,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aahle-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13.1,19.9203,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aahle-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",13.1,19.9203,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aahle-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aahle-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aahle-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aahle-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",29.4,52.3904,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aahle-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",29.4,52.3904,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aahle-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",29.4,52.3904,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aahle-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",32.8,59.1633,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aahle-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",32.8,59.1633,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aahle-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",32.8,59.1633,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aahle-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.9,7.5697,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aahle-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.9,7.5697,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aahle-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.9,7.5697,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.1,3.9841,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aahle-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.2,6.1753,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aahle-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.2,6.1753,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aahle-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",6.2,6.1753,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aahle-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.2,2.1912,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aahle-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.2,2.1912,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aahle-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.2,2.1912,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aahle-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aahle-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aahle-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.8,3.3865,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aahle-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aahle-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aahle-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aahle-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8,9.761,"percent","higher","1.5.0","excluded","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aahle-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8,9.761,"percent","higher","1.6.0","excluded","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aahle-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8,9.761,"percent","higher","1.8.0","excluded","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aahle-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aahle-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aahle-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aahle-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.3,50.1992,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aahle-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.3,50.1992,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aahle-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.3,50.1992,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aahle-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.1,49.8008,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aahle-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.1,49.8008,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aahle-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.1,49.8008,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aahle-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.1,67.7291,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aahle-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.1,67.7291,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aahle-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",37.1,67.7291,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aahle-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aahle-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aahle-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4,1.7928,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aahle-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aahle-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aahle-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aahle-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aahle-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aahle-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.3,2.3904,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aahle-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",12.8,19.3227,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aahle-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",12.8,19.3227,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aahle-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",12.8,19.3227,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aahle-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aahle-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aahle-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aahle-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aahle-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aahle-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",9.5,12.749,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aahle-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.2,14.1434,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aahle-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.2,14.1434,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aahle-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.2,14.1434,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.3,4.3825,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.3,4.3825,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",5.3,4.3825,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aahle-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.1,9.9602,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aahle-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.1,9.9602,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aahle-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.1,9.9602,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aahle-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.4,0.5976,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aahle-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.4,0.5976,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aahle-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.4,0.5976,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aahle-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.7,9.1633,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aahle-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.7,9.1633,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aahle-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7.7,9.1633,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aahle-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",20,33.6653,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aahle-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",20,33.6653,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aahle-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",20,33.6653,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aahle-2026-07-21","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.7,11.1554,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aahle-2026-07-27","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.7,11.1554,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-mini-aahle-2026-08-01","o3-mini","o3-mini","Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",8.7,11.1554,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aahle-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aahle-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aahle-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",4.1,1.992,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aahle-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.9,51.3944,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aahle-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.9,51.3944,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aahle-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",28.9,51.3944,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-07-21","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-07-27","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-08-01","qwen2-5-coder-32b-instruct","Qwen2.5 Coder 32B Instruct","Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aahle-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.1,15.9363,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aahle-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.1,15.9363,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aahle-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",11.1,15.9363,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aahle-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",22.2,38.0478,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aahle-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",22.2,38.0478,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aahle-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",22.2,38.0478,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aahle-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",27.3,48.2072,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aahle-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",27.3,48.2072,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aahle-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",27.3,48.2072,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aahle-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aahle-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aahle-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",23.4,40.4382,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aahle-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.7,33.0677,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aahle-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.7,33.0677,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aahle-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.7,33.0677,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aahle-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.1,13.9442,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aahle-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.1,13.9442,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aahle-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",10.1,13.9442,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aahle-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aahle-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aahle-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",7,7.7689,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aahle-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aahle-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aahle-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",3.8,1.3944,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aahle-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aahle-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aahle-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",19.9,33.4661,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aahle-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aahle-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aahle-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aahle-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aahle-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aahle-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","2026",14.7,23.1076,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-hle-2026-08-01","kimi-k3","Kimi K3","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","kimi-k3-max","Kimi K3 public default configuration (reasoning_effort=max; temperature=1.0)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","HLE-Full with general tools",56,84.7368,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-29","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration. The official Kimi K3 card reports 43.5 without tools and 56.0 with general tools; this carrier is the 56.0 with-tools track. Configuration metadata is attached without changing the recorded value, inclusion status, source role or underlying observation identity."],["new-model:aa-glm-5-3-flash-2026-08-27:glm-5-3-flash:humanitys-last-exam:cell:glm-5-3-flash-launch:hle-tools:0","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max effort, documented default)","glm-5-3-flash-max","GLM-5.3-Flash (max effort, documented default)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","rolling",55.3,55.3,"percent","higher","2.1.0","reference-only","direct","2026-08-26",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-26","2026-08-27","2026-08-27","source-checked","Maximum generation 163,840, 300K context and GPT-5.6 Luna medium as judge."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:hle:4","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","claude-opus-4-6-max","Claude Opus 4.6 (Max) as published in Qwen's Qwen3.8-Flash-Next comparison table","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",40,40,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Humanity's Last Exam was judged by GPT-4o. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-claude-opus-4-6-humanitys-last-exam-text-only-table","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 Max as published by Qwen","claude-opus-4-6-max","Claude Opus 4.6 Max as published by Qwen","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",40,40,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: judged by GPT-4o. Muse Glimmer 22.0 is already stored from the Muse owner card."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:hle:3","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","deepseek-v4-flash-0731-deepseek-0813-release-unspecified","DeepSeek-V4-Flash-0731 (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",33.8,33.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Humanity's Last Exam was judged by GPT-4o. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen3-6-27b-humanitys-last-exam-text-only-table","qwen3-6-27b","Qwen3.6 27B","Qwen3.6-27B as published by Qwen","qwen3-6-27b-default","Qwen3.6-27B as published by Qwen","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",24,24,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: judged by GPT-4o. Muse Glimmer 22.0 is already stored from the Muse owner card."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:hle:2","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-7-plus-unspecified","Qwen3.7-Plus (provider-published configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",34.7,34.7,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Humanity's Last Exam was judged by GPT-4o. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-7-plus-humanitys-last-exam-text-only-table","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7-Plus as published by Qwen","qwen-3-7-plus-unspecified","Qwen3.7-Plus as published by Qwen","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",34.7,34.7,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: judged by GPT-4o. Muse Glimmer 22.0 is already stored from the Muse owner card."],["provided-comparison:qwen3-8-flash-next-2026-08-26:cell:language:hle:1","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh default thinking configuration) as published in Qwen's Qwen3.8-Flash-Next comparison table","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",30.8,30.8,"percent","higher","2.1.0","excluded","direct","2026-08-26","2026-08-26","production::qwen3-8-flash-next-release-2026-08-26","qwen3-8-flash-next-release-2026-08-26","Qwen3.8-Flash-Next launch and official provider evaluations","Qwen","https://qwen.ai/blog?id=qwen3.8-flash-next","2026-08-26","2026-08-26","2026-08-26","provider-reported","Humanity's Last Exam was judged by GPT-4o. Competitor cell retained from the complete official provider table; never admitted as independent direct evidence."],["evidence-2026-08-15-qwen-3-8-27b-humanitys-last-exam-text-only","qwen-3-8-27b","Qwen3.8-27B","Qwen3.8-27B (xhigh)","qwen-3-8-27b-xhigh","Qwen3.8-27B (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only",30.8,30.8,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::qwen-3-8-27b-huggingface-2026-08-15","qwen-3-8-27b-huggingface-2026-08-15","Qwen3.8-27B official model card","Qwen","https://huggingface.co/Qwen/Qwen3.8-27B","2026-08-15","2026-08-31","2026-08-15","provider-reported","Official Qwen footnote: judged by GPT-4o. Muse Glimmer 22.0 is already stored from the Muse owner card."],["evidence-2026-08-muse-glimmer-30b-humanitys-last-exam-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64",null,"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only / no tools / 2,158 questions",22,22,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Exact text-only, no-tools result. It is not transferred to any multimodal or tool-enabled HLE configuration."],["evidence-2026-08-muse-glimmer-30b-humanitys-last-exam-high--configuration--muse-glimmer-30b-high","muse-glimmer-30b","Muse Glimmer 30B","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","muse-glimmer-30b-high","Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only / no tools / 2,158 questions",22,22,"percent","higher","1.8.0","reference-only","direct","2026-08-10","2026-08-10","production::meta-muse-glimmer-30b-methodology-2026-08-10","meta-muse-glimmer-30b-methodology-2026-08-10","Muse Glimmer Evaluation Methodology","Meta Superintelligence Lab","https://research.meta.ai/static/muse-glimmer-methodology","2026-08-10","2026-08-10","2026-08-10","provider-reported","Exact text-only, no-tools result. It is not transferred to any multimodal or tool-enabled HLE configuration."],["aa-individual:claude-fable-5:humanitys-last-exam:2026-08-27","claude-fable-5","Claude Fable 5","Claude Fable 5 (max; Artificial Analysis independent run)","claude-fable-5-max","Claude Fable 5 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",55.468025949953706,55.468025949953706,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-fable-5-2026-08-27","aa-parity-model-claude-fable-5-2026-08-27","Claude Fable 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:claude-opus-4-6-thinking:humanitys-last-exam:2026-08-29","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",39.944392956441,39.944392956441,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-claude-opus-4-6-thinking-2026-08-29","aa-current-claude-opus-4-6-thinking-2026-08-29","Claude Opus 4.6 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-6-adaptive","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:claude-opus-4-7-adaptive:humanitys-last-exam:2026-08-29","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",42.307692307692,42.307692307692,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-claude-opus-4-7-adaptive-2026-08-29","aa-current-claude-opus-4-7-adaptive-2026-08-29","Claude Opus 4.7 (Adaptive Reasoning, Max Effort) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-7","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:claude-opus-4-8:humanitys-last-exam:2026-08-27","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (max; Artificial Analysis independent run)","claude-opus-4-8-max","Claude Opus 4.8 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",48.6561631139944,48.6561631139944,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-4-8-2026-08-27","aa-parity-model-claude-opus-4-8-2026-08-27","Claude Opus 4.8 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-opus-5:humanitys-last-exam:2026-08-27","claude-opus-5","Claude Opus 5","Claude Opus 5 (max; Artificial Analysis independent run)","claude-opus-5-max","Claude Opus 5 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",54.8656163113994,54.8656163113994,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-opus-5-2026-08-27","aa-parity-model-claude-opus-5-2026-08-27","Claude Opus 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:claude-sonnet-5:humanitys-last-exam:2026-08-27","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (max; Artificial Analysis independent run)","claude-sonnet-5-max","Claude Sonnet 5 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",41.2882298424467,41.2882298424467,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-claude-sonnet-5-2026-08-27","aa-parity-model-claude-sonnet-5-2026-08-27","Claude Sonnet 5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:deepseek-v3-1:humanitys-last-exam:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","deepseek-v3-1-non-reasoning","DeepSeek V3.1 (Non-reasoning) (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",6.672845227062,6.672845227062,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-2026-08-29","aa-current-deepseek-v3-1-2026-08-29","DeepSeek V3.1 (Non-reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-current:deepseek-v3-1-reasoning:humanitys-last-exam:2026-08-29","deepseek-v3-1","DeepSeek V3.1","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","deepseek-v3-1-reasoning-default","DeepSeek V3.1 (Reasoning) (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",14.272474513438,14.272474513438,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-deepseek-v3-1-reasoning-2026-08-29","aa-current-deepseek-v3-1-reasoning-2026-08-29","DeepSeek V3.1 (Reasoning) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v3-1-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current exact independent evaluation; AA composites are excluded."],["aa-individual:deepseek-v4-flash-vision-exp:humanitys-last-exam:2026-08-27","deepseek-v4-flash-vision-exp","DeepSeek V4 Flash Vision Exp","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","deepseek-v4-flash-vision-exp-max-harness","DeepSeek V4 Flash Vision Exp (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",34.4763670064875,34.4763670064875,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-deepseek-v4-flash-vision-2026-08-27","aa-deepseek-v4-flash-vision-2026-08-27","DeepSeek V4 Flash Vision Exp individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash-vision","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual humanitys-last-exam result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-individual:deepseek-v4-pro-0813:humanitys-last-exam:2026-08-27","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","deepseek-v4-pro-0813-max","DeepSeek V4 Pro 0813 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",41.0101946246525,41.0101946246525,"percent","higher","2.1.0","reference-only","direct","2026-08-27",null,"production::aa-parity-model-deepseek-v4-pro-0813-2026-08-27","aa-parity-model-deepseek-v4-pro-0813-2026-08-27","DeepSeek V4 Pro 0813 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. AA evaluated max, while the canonical default configuration is high. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-6-flash:humanitys-last-exam:2026-08-27","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash (high; Artificial Analysis independent run)","gemini-3-6-flash-high","Gemini 3.6 Flash (high; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",40.824837812789596,40.824837812789596,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-6-flash-2026-08-27","aa-parity-model-gemini-3-6-flash-2026-08-27","Gemini 3.6 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gemini-3-7-flash:humanitys-last-exam:2026-08-27","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","gemini-3-7-flash-medium","Gemini 3.7 Flash (medium; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",38.9712696941613,38.9712696941613,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gemini-3-7-flash-2026-08-27","aa-parity-model-gemini-3-7-flash-2026-08-27","Gemini 3.7 Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-7-flash-medium","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:gemma-3-27b:humanitys-last-exam:2026-08-29","gemma-3-27b","Gemma 3 27B","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","gemma-3-27b-aa-non-reasoning-default","Gemma 3 27B Instruct (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",4.4,4.4,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-3-27b-2026-08-29","aa-current-gemma-3-27b-2026-08-29","Gemma 3 27B Instruct current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-3-27b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gemma-4-26b-a4b:humanitys-last-exam:2026-08-29","gemma-4-26b-a4b","Gemma 4 26B A4B","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","gemma-4-26b-a4b-aa-reasoning-default","Gemma 4 26B A4B (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",19.323447636701,19.323447636701,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gemma-4-26b-a4b-2026-08-29","aa-current-gemma-4-26b-a4b-2026-08-29","Gemma 4 26B A4B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gemma-4-26b-a4b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:glm-5-2:humanitys-last-exam:2026-08-29","glm-5-2","GLM-5.2","GLM-5.2 (max) (Artificial Analysis independent run)","glm-5-2-max","GLM-5.2 (max) (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",41.14921223355,41.14921223355,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-glm-5-2-2026-08-29","aa-current-glm-5-2-2026-08-29","GLM-5.2 (max) individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-2","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-individual:glm-5-3:humanitys-last-exam:2026-08-27","glm-5-3","GLM-5.3","GLM-5.3 (max; Artificial Analysis independent run)","glm-5-3-max","GLM-5.3 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",42.2613531047266,42.2613531047266,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-glm-5-3-2026-08-27","aa-parity-model-glm-5-3-2026-08-27","GLM-5.3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:glm-5-3-flash:humanitys-last-exam:2026-08-27","glm-5-3-flash","GLM-5.3-Flash","GLM-5.3-Flash (max; Artificial Analysis independent run)","glm-5-3-flash-max","GLM-5.3-Flash (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",39.8517145505097,39.8517145505097,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-glm-5-3-flash-2026-08-27","aa-glm-5-3-flash-2026-08-27","GLM-5.3-Flash individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/glm-5-3-flash","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual humanitys-last-exam result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:gpt-4-1-mini:humanitys-last-exam:2026-08-29","gpt-4-1-mini","GPT-4.1 mini","GPT-4.1 mini (Artificial Analysis completed independent run)","gpt-4-1-mini-epoch-gpt-4-1-mini-2025-04-14","GPT-4.1 mini (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",5.02,5.02,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-mini-2026-08-29","aa-current-gpt-4-1-mini-2026-08-29","GPT-4.1 mini current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-mini","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:gpt-4-1-nano:humanitys-last-exam:2026-08-29","gpt-4-1-nano","GPT-4.1 nano","GPT-4.1 nano (Artificial Analysis completed independent run)","gpt-4-1-nano-epoch-gpt-4-1-nano-2025-04-14","GPT-4.1 nano (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",3.753475440222,3.753475440222,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-gpt-4-1-nano-2026-08-29","aa-current-gpt-4-1-nano-2026-08-29","GPT-4.1 nano current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/gpt-4-1-nano","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:gpt-5-4:humanitys-last-exam:2026-08-27","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh; Artificial Analysis independent run)","gpt-5-4-xhigh","GPT-5.4 (xhigh; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",43.7442075996293,43.7442075996293,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-4-2026-08-27","aa-parity-model-gpt-5-4-2026-08-27","GPT-5.4 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-5:humanitys-last-exam:2026-08-27","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh; Artificial Analysis independent run)","gpt-5-5-xhigh","GPT-5.5 (xhigh; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",45.783132530120504,45.783132530120504,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-5-2026-08-27","aa-parity-model-gpt-5-5-2026-08-27","GPT-5.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-luna:humanitys-last-exam:2026-08-27","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max; Artificial Analysis independent run)","gpt-5-6-luna-max","GPT-5.6 Luna (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",39.4810009267841,39.4810009267841,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-luna-2026-08-27","aa-parity-model-gpt-5-6-luna-2026-08-27","GPT-5.6 Luna individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-sol:humanitys-last-exam:2026-08-27","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max; Artificial Analysis independent run)","gpt-5-6-sol-max","GPT-5.6 Sol (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",49.4902687673772,49.4902687673772,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-sol-2026-08-27","aa-parity-model-gpt-5-6-sol-2026-08-27","GPT-5.6 Sol individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:gpt-5-6-terra:humanitys-last-exam:2026-08-27","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max; Artificial Analysis independent run)","gpt-5-6-terra-max","GPT-5.6 Terra (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",42.9101019462465,42.9101019462465,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-gpt-5-6-terra-2026-08-27","aa-parity-model-gpt-5-6-terra-2026-08-27","GPT-5.6 Terra individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-5:humanitys-last-exam:2026-08-27","grok-4-5","Grok 4.5","Grok 4.5 (high; Artificial Analysis independent run)","grok-4-5-aa-2-high","Grok 4.5 (high; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",42.678405931418,42.678405931418,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-5-2026-08-27","aa-parity-model-grok-4-5-2026-08-27","Grok 4.5 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:grok-4-6:humanitys-last-exam:2026-08-27","grok-4-6","Grok 4.6","Grok 4.6 (high; Artificial Analysis independent run)","grok-4-6-high","Grok 4.6 (high; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",42.9101019462465,42.9101019462465,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-grok-4-6-2026-08-27","aa-parity-model-grok-4-6-2026-08-27","Grok 4.6 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-6","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:kimi-k2-5-reasoning:humanitys-last-exam:2026-08-29","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","kimi-k2-5-thinking","Kimi K2.5 (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",30.722891566265,30.722891566265,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-kimi-k2-5-reasoning-2026-08-29","aa-current-kimi-k2-5-reasoning-2026-08-29","Kimi K2.5 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k2-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:kimi-k3:humanitys-last-exam:2026-08-27","kimi-k3","Kimi K3","Kimi K3 (max; Artificial Analysis independent run)","kimi-k3-max","Kimi K3 (max; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",46.8952734012975,46.8952734012975,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-kimi-k3-2026-08-27","aa-parity-model-kimi-k3-2026-08-27","Kimi K3 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-current:llama-4-maverick:humanitys-last-exam:2026-08-29","llama-4-maverick","Llama 4 Maverick","Llama 4 Maverick (Artificial Analysis completed independent run)","llama-4-maverick-aa-non-reasoning-default","Llama 4 Maverick (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",4.911955514365,4.911955514365,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-maverick-2026-08-29","aa-current-llama-4-maverick-2026-08-29","Llama 4 Maverick current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-maverick","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:llama-4-scout:humanitys-last-exam:2026-08-29","llama-4-scout","Llama 4 Scout","Llama 4 Scout (Artificial Analysis completed independent run)","llama-4-scout-aa-non-reasoning-default","Llama 4 Scout (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",3.78,3.78,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-llama-4-scout-2026-08-29","aa-current-llama-4-scout-2026-08-29","Llama 4 Scout current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/llama-4-scout","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:longcat-2-0:humanitys-last-exam:2026-08-29","longcat-2-0","LongCat-2.0","LongCat 2.0 (Artificial Analysis independent run)","longcat-2-0-default","LongCat 2.0 (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",33.68860055607,33.68860055607,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-longcat-2-0-2026-08-29","aa-current-longcat-2-0-2026-08-29","LongCat 2.0 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/longcat-2-0","2026-08-29","2026-08-29","2026-08-29","independently-verified","Current Artificial Analysis individual result; exact configuration compatibility was checked independently of the composite indices."],["aa-current:mistral-medium-3-5-128b:humanitys-last-exam:2026-08-29","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Mistral Medium 3.5 (Artificial Analysis completed independent run)","mistral-medium-3-5-128b-aa-reasoning-default","Mistral Medium 3.5 (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",13.762743280816,13.762743280816,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-medium-3-5-128b-2026-08-29","aa-current-mistral-medium-3-5-128b-2026-08-29","Mistral Medium 3.5 current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:mistral-small-4-reasoning:humanitys-last-exam:2026-08-29","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","mistral-small-4-reasoning","Mistral Small 4 (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",9.870250231696,9.870250231696,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-mistral-small-4-reasoning-2026-08-29","aa-current-mistral-small-4-reasoning-2026-08-29","Mistral Small 4 (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-individual:muse-spark-1-1:humanitys-last-exam:2026-08-27","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",46.2001853568119,46.2001853568119,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-parity-model-muse-spark-1-1-2026-08-27","aa-parity-model-muse-spark-1-1-2026-08-27","Muse Spark 1.1 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently measured this individual evaluation on the current model page. The current independent AA model page matches the reviewed canonical default configuration. The composite Intelligence Index is not ingested."],["aa-individual:muse-spark-1-2:humanitys-last-exam:2026-08-27","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh; Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",45.4587581093605,45.4587581093605,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-27",null,"production::aa-muse-spark-1-2-2026-08-27","aa-muse-spark-1-2-2026-08-27","Muse Spark 1.2 individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-2","2026-08-27","2026-08-27","2026-08-27","independently-verified","Artificial Analysis independently evaluated this individual humanitys-last-exam result; the current model page supplies the value and the benchmark-specific evaluation page supplies the protocol."],["aa-current:nemotron-3-nano-30b:humanitys-last-exam:2026-08-29","nemotron-3-nano-30b","Nemotron 3 Nano 30B","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","nemotron-3-nano-30b-aa-reasoning-default","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",11.399443929564,11.399443929564,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-nemotron-3-nano-30b-2026-08-29","aa-current-nemotron-3-nano-30b-2026-08-29","NVIDIA Nemotron 3 Nano 30B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/nvidia-nemotron-3-nano-30b-a3b-reasoning","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-397b-reasoning:humanitys-last-exam:2026-08-29","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-397b-thinking","Qwen3.5 397B A17B (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",28.962001853568,28.962001853568,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-397b-reasoning-2026-08-29","aa-current-qwen3-5-397b-reasoning-2026-08-29","Qwen3.5 397B A17B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-397b-a17b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-5-122b-a10b:humanitys-last-exam:2026-08-29","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","qwen3-5-122b-a10b-aa-reasoning-default","Qwen3.5 122B A10B (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",25.208526413346,25.208526413346,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-5-122b-a10b-2026-08-29","aa-current-qwen3-5-122b-a10b-2026-08-29","Qwen3.5 122B A10B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-5-122b-a10b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen3-6-35b-a3b:humanitys-last-exam:2026-08-29","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","qwen3-6-35b-a3b-aa-reasoning-default","Qwen3.6 35B A3B (Reasoning) (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",22.24281742354,22.24281742354,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-qwen3-6-35b-a3b-2026-08-29","aa-current-qwen3-6-35b-a3b-2026-08-29","Qwen3.6 35B A3B (Reasoning) current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-6-35b-a3b","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["aa-current:qwen-3-8-flash-next:humanitys-last-exam:2026-08-29","qwen-3-8-flash-next","Qwen3.8-Flash-Next","Qwen3.8-Flash-Next (Artificial Analysis independent run)","qwen-3-8-flash-next-aa-unspecified","Qwen3.8-Flash-Next (Artificial Analysis independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",38.044485634847,38.044485634847,"percent","higher","2.1.0","reference-only","direct","2026-08-29",null,"production::aa-current-qwen-3-8-flash-next-2026-08-29","aa-current-qwen-3-8-flash-next-2026-08-29","Qwen3.8-Flash-Next individual evaluations","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-8-flash-next","2026-08-29","2026-08-29","2026-08-29","independently-verified","Artificial Analysis does not expose the reasoning-effort label needed to match the ranked default configuration."],["aa-current:trinity-large-thinking:humanitys-last-exam:2026-08-29","trinity-large-thinking","Trinity-Large-Thinking","Trinity Large Thinking (Artificial Analysis completed independent run)","trinity-large-thinking-aa-reasoning-default","Trinity Large Thinking (Artificial Analysis completed independent run)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","text-only current",15.848007414272,15.848007414272,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-29",null,"production::aa-current-trinity-large-thinking-2026-08-29","aa-current-trinity-large-thinking-2026-08-29","Trinity Large Thinking current model and individual-evaluation page","Artificial Analysis","https://artificialanalysis.ai/models/trinity-large-thinking","2026-08-29","2026-08-29","2026-08-29","independently-verified","Completed current exact individual evaluation; AA composites are excluded."],["evidence-2026-08-15-command-a-plus-humanitys-last-exam-without-tools-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",11.4,11.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-humanitys-last-exam-without-tools-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",32.3,32.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-humanitys-last-exam-without-tools-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",24.3,24.3,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-humanitys-last-exam-without-tools-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",12.8,12.8,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-humanitys-last-exam-without-tools-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",11.5,11.5,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-humanitys-last-exam-without-tools","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI","without tools",28.8,28.8,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-1932","claude-fable-5","Claude Fable 5","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort, Opus 4.8 Fallback.",null,"Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort, Opus 4.8 Fallback.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,53.3,53.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-1932--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort, Opus 4.8 Fallback.","claude-fable-5-max","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort, Opus 4.8 Fallback.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,53.3,53.3,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-056","claude-fable-5","Claude Fable 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,53.34,53.34,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-fable-5-evals","aa-claude-fable-5-evals","Artificial Analysis evaluations for claude-fable-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-fable-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-066","claude-haiku-4-5","Claude Haiku 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,4.26,4.26,"percent","higher","1.1.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-claude-4-5-haiku-evals","aa-claude-4-5-haiku-evals","Artificial Analysis evaluations for claude-4-5-haiku","Artificial Analysis","https://artificialanalysis.ai/models/claude-4-5-haiku","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-122","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,64.5,64.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1403","claude-opus-4","Claude Opus 4","Claude 4 Opus (Reasoning)",null,"Claude 4 Opus (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,11.6775,11.6775,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-opus-thinking."],["evidence-2026-07-1392","claude-opus-4-1","Claude Opus 4.1","Claude 4.1 Opus (Reasoning)",null,"Claude 4.1 Opus (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,11.8911,11.8911,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-1-opus-thinking."],["evidence-2026-07-602","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,30.8,30.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-595","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,53,53,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1934","claude-opus-4-8","Claude Opus 4.8","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort.",null,"Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,45.7,45.7,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-1934--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort.","claude-opus-4-8-max","Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,45.7,45.7,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-059","claude-opus-4-8","Claude Opus 4.8","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,45.74,45.74,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-opus-4-8-evals","aa-claude-opus-4-8-evals","Artificial Analysis evaluations for claude-opus-4-8","Artificial Analysis","https://artificialanalysis.ai/models/claude-opus-4-8","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-123","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,57.9,57.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2058","claude-opus-5","Claude Opus 5","Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.",null,"Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,64.7,64.7,"percent","higher","1.6.0","ranking-eligible","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 64.7% on HLE with tools (56.3% without tools also published)."],["evidence-2026-07-1775","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,10.2832,10.2832,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1725","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,9.5922,9.5922,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1379","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,17.2845,17.2845,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-599","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,49,49,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-062","claude-sonnet-5","Claude Sonnet 5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,39.57,39.57,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-claude-sonnet-5-evals","aa-claude-sonnet-5-evals","Artificial Analysis evaluations for claude-sonnet-5","Artificial Analysis","https://artificialanalysis.ai/models/claude-sonnet-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-124","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,57.4,57.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1633","command-a-plus","Command A+","Command A+",null,"Command A+","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,11.3531,11.3531,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1539","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,15.2456,15.2456,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1525","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,22.2428,22.2428,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-081","deepseek-v4-flash","DeepSeek V4 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,32.07,32.07,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-flash-evals","aa-deepseek-v4-flash-evals","Artificial Analysis evaluations for deepseek-v4-flash","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-flash","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-132","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,8.1,8.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-078","deepseek-v4-pro","DeepSeek V4 Pro","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35.87,35.87,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-deepseek-v4-pro-evals","aa-deepseek-v4-pro-evals","Artificial Analysis evaluations for deepseek-v4-pro","Artificial Analysis","https://artificialanalysis.ai/models/deepseek-v4-pro","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-133","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,7.7,7.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1837","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,12.7433,12.7433,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1799","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,21.0843,21.0843,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-070","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,44.72,44.72,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gemini-3-1-pro-preview-evals","aa-gemini-3-1-pro-preview-evals","Artificial Analysis evaluations for gemini-3-1-pro-preview","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-1-pro-preview","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-017","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,44.4,44.4,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-115","gemini-3-5-flash","Gemini 3.5 Flash","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,39.85,39.85,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gemini-3-5-flash-medium-evals","aa-gemini-3-5-flash-medium-evals","Artificial Analysis evaluations for gemini-3-5-flash-medium","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-medium","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-008","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,40.2,40.2,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-128","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,40.2,40.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2040","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis Humanity's Last Exam independent evaluation.",null,"Artificial Analysis Humanity's Last Exam independent evaluation.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,17.5,17.5,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA HLE 17.5%."],["evidence-2026-07-2000","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis Humanity's Last Exam independent evaluation.",null,"Artificial Analysis Humanity's Last Exam independent evaluation.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,38.3,38.3,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA HLE 38.3%."],["evidence-2026-07-1881","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,14.7822,14.7822,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1620","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,18.2576,18.2576,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-604","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.5,26.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1740","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,13.3457,13.3457,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1553","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,25.1158,25.1158,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-597","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,50.4,50.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-596","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,52.3,52.3,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-594","glm-5-2","GLM-5.2","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,54.7,54.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1339","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.506,26.506,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1339--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.506,26.506,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1324","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,25.5792,25.5792,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1324--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,25.5792,25.5792,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1351","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,14.5968,14.5968,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1351--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,14.5968,14.5968,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1303","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35.4495,35.4495,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1303--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35.4495,35.4495,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1314","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,33.4569,33.4569,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1314--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,33.4569,33.4569,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-104","gpt-5-3-codex","GPT-5.3-Codex","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,39.9,39.9,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-3-codex-evals","aa-gpt-5-3-codex-evals","Artificial Analysis evaluations for gpt-5-3-codex","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-3-codex","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-100","gpt-5-4","GPT-5.4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,41.61,41.61,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-4-evals","aa-gpt-5-4-evals","Artificial Analysis evaluations for gpt-5-4","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-126","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,52.1,52.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1293","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.645,26.645,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1293--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.645,26.645,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-601","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,37.7,37.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-111","gpt-5-5","GPT-5.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,44.3,44.3,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-5-evals","aa-gpt-5-5-evals","Artificial Analysis evaluations for gpt-5-5","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-027","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,41.4,41.4,"percent","higher","1.0.0","ranking-eligible","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-125","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,52.2,52.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-593","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpt-5-5-pro-default-high","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,57.2,57.2,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-052","gpt-5-6-luna","GPT-5.6 Luna","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,37.21,37.21,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-luna-evals","aa-gpt-5-6-luna-evals","Artificial Analysis evaluations for gpt-5-6-luna","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-luna","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1933","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis HLE evaluation; configuration max.",null,"Artificial Analysis HLE evaluation; configuration max.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,47.2,47.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-1933--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis HLE evaluation; configuration max.","gpt-5-6-sol-max","Artificial Analysis HLE evaluation; configuration max.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,47.2,47.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","From Artificial Analysis Humanity's Last Exam leaderboard page summary checked 2026-07-16."],["evidence-2026-07-044","gpt-5-6-sol","GPT-5.6 Sol","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,47.17,47.17,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-sol-evals","aa-gpt-5-6-sol-evals","Artificial Analysis evaluations for gpt-5-6-sol","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-sol","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-048","gpt-5-6-terra","GPT-5.6 Terra","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,41.8,41.8,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-gpt-5-6-terra-evals","aa-gpt-5-6-terra-evals","Artificial Analysis evaluations for gpt-5-6-terra","Artificial Analysis","https://artificialanalysis.ai/models/gpt-5-6-terra","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1870","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,11.0656,11.0656,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1658","grok-4","Grok 4","Grok 4",null,"Grok 4","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,23.911,23.911,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1762","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,16.9601,16.9601,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1447","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,32.2057,32.2057,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-074","grok-4-3","Grok 4.3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35.03,35.03,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-3-evals","aa-grok-4-3-evals","Artificial Analysis evaluations for grok-4-3","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-129","grok-4-3","Grok 4.3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35,35,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-118","grok-4-5","Grok 4.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,40.27,40.27,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-grok-4-5-evals","aa-grok-4-5-evals","Artificial Analysis evaluations for grok-4-5","Artificial Analysis","https://artificialanalysis.ai/models/grok-4-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-1892","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,7.4606,7.4606,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1901","hy3","Hy3","tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.",null,"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,53.2,53.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-06","2026-07-06","production::tencent-hy3-hf","tencent-hy3-hf","Tencent Hy3 model card on Hugging Face","Tencent Hy Team","https://huggingface.co/tencent/Hy3","2026-07-06","2026-07-16","2026-07-16","provider-reported","Humanity's Last Exam score listed in Hugging Face evaluation results for tencent/Hy3; may include tool settings—retain as provider-attached."],["evidence-2026-07-1849","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,6.3485,6.3485,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1670","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,22.3355,22.3355,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-131","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,30.1,30.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1421","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,35.9129,35.9129,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1437","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,32.7618,32.7618,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1941","kimi-k3","Kimi K3","HLE-Full with tools; max reasoning; provider-published Kimi K3 evaluation.",null,"HLE-Full with tools; max reasoning; provider-published Kimi K3 evaluation.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,56,56,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 56.0% on HLE-Full with tools; 43.5% without tools also published."],["evidence-2026-07-1941--configuration--kimi-k3-max","kimi-k3","Kimi K3","HLE-Full with tools; max reasoning; provider-published Kimi K3 evaluation.","kimi-k3-max","HLE-Full with tools; max reasoning; provider-published Kimi K3 evaluation.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,56,56,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 56.0% on HLE-Full with tools; 43.5% without tools also published."],["evidence-2026-07-1787","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,8.2484,8.2484,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1589","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,21.1307,21.1307,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1576","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,25.1622,25.1622,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-600","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,48,48,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1751","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,12.4652,12.4652,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1690","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,22.1965,22.1965,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1564","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,19.1381,19.1381,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-107","minimax-m3","MiniMax M3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,37.12,37.12,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-minimax-m3-evals","aa-minimax-m3-evals","Artificial Analysis evaluations for minimax-m3","Artificial Analysis","https://artificialanalysis.ai/models/minimax-m3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-092","mistral-large-3","Mistral Large 3","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,4.08,4.08,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-large-3-evals","aa-mistral-large-3-evals","Artificial Analysis evaluations for mistral-large-3","Artificial Analysis","https://artificialanalysis.ai/models/mistral-large-3","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-084","mistral-medium-3-5","Mistral Medium 3.5","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,12.79,12.79,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-medium-3-5-evals","aa-mistral-medium-3-5-evals","Artificial Analysis evaluations for mistral-medium-3-5","Artificial Analysis","https://artificialanalysis.ai/models/mistral-medium-3-5","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-088","mistral-small-4","Mistral Small 4","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,9.5,9.5,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-mistral-small-4-evals","aa-mistral-small-4-evals","Artificial Analysis evaluations for mistral-small-4","Artificial Analysis","https://artificialanalysis.ai/models/mistral-small-4","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-598","muse-spark","Muse Spark","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,50.4,50.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-592","muse-spark-1-1","Muse Spark 1.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,62.1,62.1,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1917","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,45.1,45.1,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1917--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,45.1,45.1,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::aa-hle-leaderboard","aa-hle-leaderboard","Humanity's Last Exam leaderboard (AA)","Artificial Analysis","https://artificialanalysis.ai/evaluations/humanitys-last-exam","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1911","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Humanity's Last Exam (with tools as published)).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Humanity's Last Exam (with tools as published)).","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,62.1,62.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Humanity's Last Exam (with tools as published); retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1911--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Humanity's Last Exam (with tools as published)).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Humanity's Last Exam (with tools as published)).","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,62.1,62.1,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Humanity's Last Exam (with tools as published); retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1828","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,19.1844,19.1844,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1606","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.5524,26.5524,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1646","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,8.9435,8.9435,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["evidence-2026-07-1860","o1","o1","o1",null,"o1","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,7.7496,7.7496,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1364","o3","o3","o3",null,"o3","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,20.0447,20.0447,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1811","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,17.5112,17.5112,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1811--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,17.5112,17.5112,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1681","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,26.1816,26.1816,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1500","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,23.4476,23.4476,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1511","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,22.1965,22.1965,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1712","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,19.7405,19.7405,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1486","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,27.2938,27.2938,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1701","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,13.9018,13.9018,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1460","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,21.5941,21.5941,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1470","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,28.8693,28.8693,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-models-leaderboard","aa-models-leaderboard","Artificial Analysis Models Leaderboard","Artificial Analysis","https://artificialanalysis.ai/leaderboards/models","2026-07-15","2026-07-16","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-603","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,28.8,28.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-097","qwen-3-7-max","Qwen3.7-Max","Artificial Analysis public model evaluation page; exact provider variant as listed on the page.",null,"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,38.09,38.09,"percent","higher","1.1.0","reference-only","direct","2026-07-15","2026-07-15","production::aa-qwen3-7-max-evals","aa-qwen3-7-max-evals","Artificial Analysis evaluations for qwen3-7-max","Artificial Analysis","https://artificialanalysis.ai/models/qwen3-7-max","2026-07-15","2026-07-15","2026-07-15","source-checked","Extracted from Artificial Analysis public model page structured evaluation fields. Independent-lab provenance."],["evidence-2026-07-127","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,41.4,41.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-130","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,34.7,34.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-08-qwen-3-8-max-hle-no-tools","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,43.6,43.6,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen no-tools result: 43.6 on Humanity's Last Exam. The separate tools-enabled 56.2 result is not mixed into this benchmark row."],["evidence-2026-08-qwen-3-8-max-hle-no-tools--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","humanitys-last-exam","Humanity's Last Exam","reasoning","Center for AI Safety and Scale AI",null,43.6,43.6,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen no-tools result: 43.6 on Humanity's Last Exam. The separate tools-enabled 56.2 result is not mixed into this benchmark row."],["evidence-livebench-2026-06-25-a2236522a3e20a80-reasoning","claude-fable-5","Claude Fable 5","claude-fable-5-max-effort (max)","claude-fable-5-max","claude-fable-5-max-effort (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",89.65375,89.65375,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":89.65375,\"Coding\":85.992,\"Agentic Coding\":62.171667,\"Mathematics\":95.98575,\"Data Analysis\":80.537667,\"Language\":90.684,\"IF\":75.77075}."],["evidence-livebench-2026-06-25-a47b21391d7c8836-reasoning","claude-opus-4-5","Claude Opus 4.5","claude-opus-4-5-20251101-thinking-64k-high-effort (high)","claude-opus-4-5-livebench-2026-06-25-high","claude-opus-4-5-20251101-thinking-64k-high-effort (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",80.0865,80.0865,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":80.0865,\"Coding\":79.654,\"Agentic Coding\":39.697,\"Mathematics\":90.389,\"Data Analysis\":74.441667,\"Language\":81.261,\"IF\":62.546}."],["evidence-livebench-2026-06-25-26851d69af637ca5-reasoning","claude-opus-4-6","Claude Opus 4.6","claude-opus-4-6-thinking-auto-high-effort (high)","claude-opus-4-6-livebench-2026-06-25-high","claude-opus-4-6-thinking-auto-high-effort (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",88.673,88.673,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":88.673,\"Coding\":78.1845,\"Agentic Coding\":48.989667,\"Mathematics\":89.317,\"Data Analysis\":69.893,\"Language\":83.269667,\"IF\":63.3125}."],["evidence-livebench-2026-06-25-559c84d4f96c910f-reasoning","claude-opus-4-7","Claude Opus 4.7","claude-opus-4-7-xhigh-effort (xhigh)","claude-opus-4-7-livebench-2026-06-25-xhigh","claude-opus-4-7-xhigh-effort (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",87.19225,87.19225,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":87.19225,\"Coding\":82.088,\"Agentic Coding\":50.656333,\"Mathematics\":92.85375,\"Data Analysis\":78.263667,\"Language\":77.914,\"IF\":66.7375}."],["evidence-livebench-2026-06-25-117160939ec5d74e-reasoning","claude-opus-4-8","Claude Opus 4.8","claude-opus-4-8-max-effort (max)","claude-opus-4-8-max","claude-opus-4-8-max-effort (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",89.19225,89.19225,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":89.19225,\"Coding\":81.828,\"Agentic Coding\":50.505,\"Mathematics\":94.3165,\"Data Analysis\":66.033333,\"Language\":79.659333,\"IF\":72.0335}."],["evidence-livebench-2026-06-25-029271c8c3a329de-reasoning","claude-opus-5","Claude Opus 5","claude-opus-5-max-effort (max)","claude-opus-5-max","claude-opus-5-max-effort (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",91.2115,91.2115,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":91.2115,\"Coding\":81.4455,\"Agentic Coding\":65.202,\"Mathematics\":95.731,\"Data Analysis\":74.550333,\"Language\":88.688,\"IF\":63.7665}."],["evidence-livebench-2026-06-25-e51bbd5ee0b830c9-reasoning","claude-sonnet-4-6","Claude Sonnet 4.6","claude-sonnet-4-6-thinking-auto-medium-effort (medium)","claude-sonnet-4-6-livebench-2026-06-25-medium","claude-sonnet-4-6-thinking-auto-medium-effort (medium)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",84.76925,84.76925,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":84.76925,\"Coding\":79.2715,\"Agentic Coding\":42.626,\"Mathematics\":86.994,\"Data Analysis\":77.946,\"Language\":76.102667,\"IF\":63.22075}."],["evidence-livebench-2026-06-25-7fa6740513a3cb75-reasoning","claude-sonnet-5","Claude Sonnet 5","claude-sonnet-5-xhigh-effort (xhigh)","claude-sonnet-5-xhigh","claude-sonnet-5-xhigh-effort (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",88.69225,88.69225,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":88.69225,\"Coding\":80.68,\"Agentic Coding\":59.394,\"Mathematics\":92.94225,\"Data Analysis\":71.740667,\"Language\":74.970333,\"IF\":63.8585}."],["evidence-livebench-2026-06-25-f61ff5cf8e1cc88d-reasoning","deepseek-v4-flash","DeepSeek V4 Flash","deepseek-v4-flash (reasoning configuration not stated by LiveBench)","deepseek-v4-flash-livebench-2026-06-25-unspecified","deepseek-v4-flash (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",70.58175,70.58175,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":70.58175,\"Coding\":69.228,\"Agentic Coding\":37.626,\"Mathematics\":79.6475,\"Data Analysis\":68.023,\"Language\":70.124,\"IF\":63.1375}."],["evidence-livebench-2026-06-25-52f67337446f9453-reasoning","deepseek-v4-flash-0731","DeepSeek V4 Flash 0731","deepseek-v4-flash-0731 (reasoning configuration not stated by LiveBench)","deepseek-v4-flash-0731-livebench-2026-06-25-unspecified","deepseek-v4-flash-0731 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",86.6345,86.6345,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":86.6345,\"Coding\":74.9845,\"Agentic Coding\":46.767667,\"Mathematics\":86.78975,\"Data Analysis\":79.327333,\"Language\":79.175,\"IF\":65.51675}."],["evidence-livebench-2026-06-25-052e4fc0bc88e2a5-reasoning","deepseek-v4-pro","DeepSeek V4 Pro","deepseek-v4-pro (reasoning configuration not stated by LiveBench)","deepseek-v4-pro-livebench-2026-06-25-unspecified","deepseek-v4-pro (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",82.69225,82.69225,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":82.69225,\"Coding\":69.994,\"Agentic Coding\":42.626,\"Mathematics\":90.6755,\"Data Analysis\":74.537667,\"Language\":78.130333,\"IF\":62.35}."],["evidence-livebench-2026-06-25-16e962c0a4607e06-reasoning","deepseek-v4-pro-0813","DeepSeek V4 Pro 0813","deepseek-v4-pro-0813 (reasoning configuration not stated by LiveBench)","deepseek-v4-pro-0813-livebench-2026-06-25-unspecified","deepseek-v4-pro-0813 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",85.84125,85.84125,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":85.84125,\"Coding\":77.1585,\"Agentic Coding\":54.949667,\"Mathematics\":95.08625,\"Data Analysis\":79.243333,\"Language\":82.074,\"IF\":67.7}."],["evidence-livebench-2026-06-25-83b7b461ba628c27-reasoning","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","gemini-3.1-pro-preview-high (high)","gemini-3-1-pro-preview-livebench-2026-06-25-high","gemini-3.1-pro-preview-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",84.00475,84.00475,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":84.00475,\"Coding\":76.4545,\"Agentic Coding\":44.141333,\"Mathematics\":91.045,\"Data Analysis\":78.541333,\"Language\":85.376,\"IF\":79.1}."],["evidence-livebench-2026-06-25-b047c6ecedbb70e1-reasoning","gemini-3-5-flash","Gemini 3.5 Flash","gemini-3.5-flash-high (high)","gemini-3-5-flash-high","gemini-3.5-flash-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",82.00475,82.00475,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":82.00475,\"Coding\":78.1845,\"Agentic Coding\":48.989667,\"Mathematics\":88.24375,\"Data Analysis\":64.857333,\"Language\":84.584333,\"IF\":75.6}."],["evidence-livebench-2026-06-25-9357096bc975fd92-reasoning","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite-high (high)","gemini-3-5-flash-lite-livebench-2026-06-25-high","gemini-3.5-flash-lite-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",60.1875,60.1875,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":60.1875,\"Coding\":76.0715,\"Agentic Coding\":45.252667,\"Mathematics\":73.7395,\"Data Analysis\":53.249667,\"Language\":71.820333,\"IF\":67.23775}."],["evidence-livebench-2026-06-25-a33b689137690edc-reasoning","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash-high (high)","gemini-3-6-flash-high","gemini-3.6-flash-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",85.149,85.149,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":85.149,\"Coding\":77.863,\"Agentic Coding\":43.434333,\"Mathematics\":86.4025,\"Data Analysis\":62.999333,\"Language\":83.897667,\"IF\":75.36675}."],["evidence-livebench-2026-06-25-0acab27725298752-reasoning","gemini-3-7-flash","Gemini 3.7 Flash","gemini-3.7-flash-high (high)","gemini-3-7-flash-high","gemini-3.7-flash-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",87.798,87.798,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":87.798,\"Coding\":78.8885,\"Agentic Coding\":58.283,\"Mathematics\":93.468,\"Data Analysis\":67.964333,\"Language\":85.455,\"IF\":79.92525}."],["evidence-livebench-2026-06-25-bb43a640f04c8e55-reasoning","glm-5-2","GLM-5.2","glm-5.2 (reasoning configuration not stated by LiveBench)","glm-5-2-livebench-2026-06-25-unspecified","glm-5.2 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",78.625,78.625,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":78.625,\"Coding\":79.654,\"Agentic Coding\":51.767667,\"Mathematics\":89.78125,\"Data Analysis\":73.739667,\"Language\":76.241667,\"IF\":62.29175}."],["evidence-livebench-2026-06-25-52883613b6c12c4a-reasoning","gpt-5-2","GPT-5.2","gpt-5.2-2025-12-11-high (high)","gpt-5-2-livebench-2026-06-25-high","gpt-5.2-2025-12-11-high (high)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",83.2115,83.2115,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":83.2115,\"Coding\":76.0715,\"Agentic Coding\":50.252667,\"Mathematics\":93.166,\"Data Analysis\":78.163333,\"Language\":79.809,\"IF\":61.77075}."],["evidence-livebench-2026-06-25-8ebfb0eb99fdb07f-reasoning","gpt-5-2-codex","GPT-5.2 Codex","gpt-5.2-codex (reasoning configuration not stated by LiveBench)","gpt-5-2-codex-livebench-2026-06-25-unspecified","gpt-5.2-codex (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",77.7115,77.7115,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":77.7115,\"Coding\":83.6195,\"Agentic Coding\":49.394,\"Mathematics\":88.77375,\"Data Analysis\":78.204,\"Language\":73.678333,\"IF\":66.44575}."],["evidence-livebench-2026-06-25-2f7f719fd1759034-reasoning","gpt-5-4","GPT-5.4","gpt-5.4-xhigh (xhigh)","gpt-5-4-xhigh","gpt-5.4-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",88.1155,88.1155,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":88.1155,\"Coding\":77.5415,\"Agentic Coding\":53.838333,\"Mathematics\":94.148,\"Data Analysis\":79.313333,\"Language\":82.633667,\"IF\":70.21675}."],["evidence-livebench-2026-06-25-8cb7dac7b8aa39e1-reasoning","gpt-5-4-mini","GPT-5.4 mini","gpt-5.4-mini-xhigh (xhigh)","gpt-5-4-mini-xhigh","gpt-5.4-mini-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",71.32375,71.32375,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":71.32375,\"Coding\":71.624,\"Agentic Coding\":41.666667,\"Mathematics\":78.462,\"Data Analysis\":70.785667,\"Language\":70.951667,\"IF\":59.804}."],["evidence-livebench-2026-06-25-2f4ad7a6e1ad2bd8-reasoning","gpt-5-4-nano","GPT-5.4 nano","gpt-5.4-nano-xhigh (xhigh)","gpt-5-4-nano-xhigh","gpt-5.4-nano-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",81.097,81.097,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":81.097,\"Coding\":70.8385,\"Agentic Coding\":46.767667,\"Mathematics\":90.977,\"Data Analysis\":67.642667,\"Language\":62.507,\"IF\":67.20475}."],["evidence-livebench-2026-06-25-333efa32f8582c4f-reasoning","gpt-5-5","GPT-5.5","gpt-5.5-xhigh (xhigh)","gpt-5-5-xhigh","gpt-5.5-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",89.65375,89.65375,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":89.65375,\"Coding\":82.1495,\"Agentic Coding\":53.989667,\"Mathematics\":95.85725,\"Data Analysis\":81.575667,\"Language\":87.363,\"IF\":70.72925}."],["evidence-livebench-2026-06-25-fbc0ae889c0c1ed2-reasoning","gpt-5-6-luna","GPT-5.6 Luna","gpt-5.6-luna-max (max)","gpt-5-6-luna-max","gpt-5.6-luna-max (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",85.64425,85.64425,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":85.64425,\"Coding\":82.915,\"Agentic Coding\":48.434333,\"Mathematics\":87.20075,\"Data Analysis\":78.033333,\"Language\":72.567,\"IF\":60.121}."],["evidence-livebench-2026-06-25-c01bd6da585ec523-reasoning","gpt-5-6-sol","GPT-5.6 Sol","gpt-5.6-sol-max (max)","gpt-5-6-sol-max","gpt-5.6-sol-max (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",91.65375,91.65375,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":91.65375,\"Coding\":83.941,\"Agentic Coding\":56.212,\"Mathematics\":96.19775,\"Data Analysis\":79.840667,\"Language\":87.683667,\"IF\":71.846}."],["evidence-livebench-2026-06-25-69de2a39a61a5be5-reasoning","gpt-5-6-terra","GPT-5.6 Terra","gpt-5.6-terra-max (max)","gpt-5-6-terra-max","gpt-5.6-terra-max (max)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",90.6345,90.6345,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":90.6345,\"Coding\":78.2455,\"Agentic Coding\":54.949667,\"Mathematics\":94.90825,\"Data Analysis\":79.306667,\"Language\":82.892667,\"IF\":64.6165}."],["evidence-livebench-2026-06-25-5dcbcd9086871d59-reasoning","grok-4-3","Grok 4.3","grok-4.3 (reasoning configuration not stated by LiveBench)","grok-4-3-livebench-2026-06-25-unspecified","grok-4.3 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",70.822,70.822,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":70.822,\"Coding\":69.9325,\"Agentic Coding\":18.535333,\"Mathematics\":84.339,\"Data Analysis\":55.772667,\"Language\":73.58,\"IF\":62.75}."],["evidence-livebench-2026-06-25-d3b23c79491a23d8-reasoning","grok-4-5","Grok 4.5","grok-4.5 (reasoning configuration not stated by LiveBench)","grok-4-5-livebench-2026-06-25-unspecified","grok-4.5 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",87.173,87.173,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":87.173,\"Coding\":68.5855,\"Agentic Coding\":56.464667,\"Mathematics\":90.8245,\"Data Analysis\":73.037667,\"Language\":82.795333,\"IF\":71.52925}."],["evidence-livebench-2026-06-25-44968622fb48d054-reasoning","grok-4-6","Grok 4.6","grok-4.6 (reasoning configuration not stated by LiveBench)","grok-4-6-livebench-2026-06-25-unspecified","grok-4.6 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",90.5095,90.5095,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":90.5095,\"Coding\":76.776,\"Agentic Coding\":57.02,\"Mathematics\":92.568,\"Data Analysis\":73.857,\"Language\":83.697,\"IF\":71.8665}."],["evidence-livebench-2026-06-25-9e66b4a1830c1888-reasoning","grok-build-0-1","Grok Build 0.1","grok-build-0.1 (reasoning configuration not stated by LiveBench)","grok-build-0-1-livebench-2026-06-25-unspecified","grok-build-0.1 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",76.37025,76.37025,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":76.37025,\"Coding\":65.3855,\"Agentic Coding\":45.808,\"Mathematics\":78.42875,\"Data Analysis\":70.794,\"Language\":72.46,\"IF\":65.22075}."],["evidence-livebench-2026-06-25-522da15c917fc2e1-reasoning","inkling","Inkling","inkling-xhigh (xhigh)","inkling-xhigh","inkling-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",78.34625,78.34625,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":78.34625,\"Coding\":71.0195,\"Agentic Coding\":49.394,\"Mathematics\":88.36475,\"Data Analysis\":72.778,\"Language\":73.458667,\"IF\":70.09575}."],["evidence-livebench-2026-06-25-1705c6f49b4f4c8d-reasoning","kimi-k2-6","Kimi K2.6","kimi-k2.6-thinking (reasoning configuration not stated by LiveBench)","kimi-k2-6-livebench-2026-06-25-unspecified","kimi-k2.6-thinking (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",79.375,79.375,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":79.375,\"Coding\":78.567,\"Agentic Coding\":46.919333,\"Mathematics\":84.27625,\"Data Analysis\":65.134333,\"Language\":75.141333,\"IF\":64.35825}."],["evidence-livebench-2026-06-25-fcab983a97e91c0d-reasoning","kimi-k2-7-code","Kimi K2.7 Code","kimi-k2.7-code (reasoning configuration not stated by LiveBench)","kimi-k2-7-code-livebench-2026-06-25-unspecified","kimi-k2.7-code (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",82.80775,82.80775,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":82.80775,\"Coding\":73.959,\"Agentic Coding\":45.656333,\"Mathematics\":79.60425,\"Data Analysis\":62.657333,\"Language\":77.905667,\"IF\":56.29175}."],["evidence-livebench-2026-06-25-80a8bf700c991501-reasoning","kimi-k3","Kimi K3","kimi-k3 (reasoning configuration not stated by LiveBench)","kimi-k3-livebench-2026-06-25-unspecified","kimi-k3 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",90.673,90.673,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":90.673,\"Coding\":81.4455,\"Agentic Coding\":62.171667,\"Mathematics\":84.43675,\"Data Analysis\":78.733667,\"Language\":85.528,\"IF\":71.36275}."],["evidence-livebench-2026-06-25-13b76ebe52da5091-reasoning","minimax-m3","MiniMax M3","minimax-m3 (reasoning configuration not stated by LiveBench)","minimax-m3-livebench-2026-06-25-unspecified","minimax-m3 (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",74.48075,74.48075,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":74.48075,\"Coding\":68.2025,\"Agentic Coding\":40.656333,\"Mathematics\":76.9475,\"Data Analysis\":76.166333,\"Language\":76.835667,\"IF\":57.50825}."],["evidence-livebench-2026-06-25-75fa3563b2f128e7-reasoning","muse-spark-1-1","Muse Spark 1.1","muse-spark-1.1-xhigh (xhigh)","muse-spark-1-1-xhigh","muse-spark-1.1-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",87.73075,87.73075,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":87.73075,\"Coding\":77.1585,\"Agentic Coding\":58.535333,\"Mathematics\":87.14475,\"Data Analysis\":72.548,\"Language\":74.342,\"IF\":69.6375}."],["evidence-livebench-2026-06-25-a92908a4d1c21cd8-reasoning","muse-spark-1-2","Muse Spark 1.2","muse-spark-1.2-xhigh (xhigh)","muse-spark-1-2-xhigh","muse-spark-1.2-xhigh (xhigh)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",90.00475,90.00475,"percent","higher","2.0.0","ranking-eligible","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":90.00475,\"Coding\":77.5415,\"Agentic Coding\":57.575667,\"Mathematics\":91.2035,\"Data Analysis\":76.458333,\"Language\":78.574667,\"IF\":74.325}."],["evidence-livebench-2026-06-25-802b617845602e2b-reasoning","qwen3-6-27b","Qwen3.6 27B","qwen3.6-27b (reasoning configuration not stated by LiveBench)","qwen3-6-27b-livebench-2026-06-25-unspecified","qwen3.6-27b (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",70.28375,70.28375,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":70.28375,\"Coding\":71.785,\"Agentic Coding\":39.292667,\"Mathematics\":79.8685,\"Data Analysis\":70.426667,\"Language\":63.304333,\"IF\":53.229}."],["evidence-livebench-2026-06-25-60e888230a7cb7fb-reasoning","qwen3-6-plus","Qwen3.6 Plus","qwen3.6-plus (reasoning configuration not stated by LiveBench)","qwen3-6-plus-livebench-2026-06-25-unspecified","qwen3.6-plus (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",75.827,75.827,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":75.827,\"Coding\":78.1845,\"Agentic Coding\":41.363667,\"Mathematics\":83.72475,\"Data Analysis\":69.911667,\"Language\":74.989333,\"IF\":58.342}."],["evidence-livebench-2026-06-25-7f8cac68e673710d-reasoning","qwen-3-7-max","Qwen3.7-Max","qwen3.7-max (reasoning configuration not stated by LiveBench)","qwen-3-7-max-livebench-2026-06-25-unspecified","qwen3.7-max (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",83.3365,83.3365,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":83.3365,\"Coding\":74.219,\"Agentic Coding\":43.586,\"Mathematics\":85.24825,\"Data Analysis\":71.786,\"Language\":79.739,\"IF\":74.0415}."],["evidence-livebench-2026-06-25-bc33b81a67813da0-reasoning","qwen-3-8-max","Qwen3.8 Max","qwen3.8-max (reasoning configuration not stated by LiveBench)","qwen-3-8-max-livebench-2026-06-25-unspecified","qwen3.8-max (reasoning configuration not stated by LiveBench)","livebench","LiveBench","reasoning","LiveBench team","2026-06-25",88.2115,88.2115,"percent","higher","2.0.0","reference-only","direct","2026-08-15",null,"production::refresh-livebench","refresh-livebench","LiveBench 2026-06-25 complete owner table","LiveBench","https://livebench.ai/","2026-08-15","2026-08-15","2026-08-15","official-leaderboard","Complete 43-row owner table captured. All category scores retained in the run extraction artifact: {\"Reasoning\":88.2115,\"Coding\":72.872,\"Agentic Coding\":64.646667,\"Mathematics\":91.31225,\"Data Analysis\":78.413,\"Language\":79.687333,\"IF\":74.0835}."],["evidence-2026-07-855","claude-fable-5","Claude Fable 5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,78.31,78.31,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-859","claude-opus-4-5","Claude Opus 4.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,75.96,75.96,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-858","claude-opus-4-6","Claude Opus 4.6","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,76.33,76.33,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-857","claude-opus-4-7","Claude Opus 4.7","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,76.91,76.91,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-856","claude-opus-4-8","Claude Opus 4.8","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,77.22,77.22,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-860","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,75.47,75.47,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-854","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,79.93,79.93,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-861","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,75.02,75.02,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-865","glm-5-1","GLM-5.1","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,70.18,70.18,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-863","gpt-5-3-codex","GPT-5.3-Codex","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,72.76,72.76,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-853","gpt-5-4","GPT-5.4","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,80.28,80.28,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-866","gpt-5-4-nano","GPT-5.4 nano","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,70.13,70.13,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-852","gpt-5-5","GPT-5.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,80.71,80.71,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-868","kimi-k2-5","Kimi K2.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,69.07,69.07,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-867","minimax-m3","MiniMax M3","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,70.02,70.02,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-864","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,70.85,70.85,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["evidence-2026-07-862","qwen-3-7-max","Qwen3.7-Max","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","livebench","LiveBench","reasoning","LiveBench team",null,74.29,74.29,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::llm-stats-livebench","llm-stats-livebench","LiveBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/livebench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants. Retained as historical corroboration only: this unversioned aggregator observation is superseded for ranking by the complete versioned LiveBench owner table."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-mrcr1m-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",76.9,88.4007,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mrcr1m-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",37.5,19.1564,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mrcr1m-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",37.5,19.1564,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-mrcr1m-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",37.5,19.1564,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",78.7,91.5641,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-mrcr1m-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.3,99.6485,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mrcr1m-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",44.7,31.8102,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mrcr1m-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",44.7,31.8102,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-mrcr1m-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",44.7,31.8102,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",83.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mrcr1m-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",26.6,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mrcr1m-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",26.6,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mrcr1m-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",26.6,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mrcr1m-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",54.1,48.3304,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mrcr1m-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",54.1,48.3304,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-mrcr1m-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcr1m","MRCR 1M","reasoning","DeepSeek-AI","2026",54.1,48.3304,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mrcrv2-128-256-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",59.2,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mrcrv2-128-256-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",59.2,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-mrcrv2-128-256-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",59.2,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",87.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",87.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-128-256","OpenAI MRCR v2 8-needle 128K-256K","reasoning","OpenAI","2026",87.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-64-128","OpenAI MRCR v2 8-needle 64K-128K","reasoning","OpenAI","2026",83.1,50,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-64-128","OpenAI MRCR v2 8-needle 64K-128K","reasoning","OpenAI","2026",83.1,50,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","benchlm-mrcrv2-64-128","OpenAI MRCR v2 8-needle 64K-128K","reasoning","OpenAI","2026",83.1,50,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-780","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,70.6,70.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-775","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,95,95,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-776","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,95,95,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-782","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,66.8,66.8,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-781","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,69.2,69.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-778","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,71.6,71.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-777","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,73.6,73.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-779","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","super-gpqa","SuperGPQA","reasoning","SuperGPQA authors",null,71.4,71.4,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["benchlm-ref-deepseek-v4-flash-base-winogrande-2026-07-21","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",79.5,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-winogrande-2026-07-27","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",79.5,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-base-winogrande-2026-08-01","deepseek-v4-flash-base","DeepSeek V4 Flash Base","Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",79.5,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-winogrande-2026-07-21","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",81.5,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-winogrande-2026-07-27","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",81.5,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-base-winogrande-2026-08-01","deepseek-v4-pro-base","DeepSeek V4 Pro Base","Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","benchlm-winogrande","WinoGrande","reasoning","DeepSeek-AI","2026",81.5,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-07-21","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-07-27","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-08-01","claude-3-haiku","Claude 3 Haiku","Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-07-21","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.2,61.2245,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-07-27","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.2,61.2245,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-08-01","claude-4-sonnet","Claude 4 Sonnet","Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.2,61.2245,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaomniscienceindex-2026-07-21","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",40.2,100,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaomniscienceindex-2026-07-27","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",40.2,100,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-fable-aaomniscienceindex-2026-08-01","claude-fable-5","Claude Fable 5","Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",40.2,100,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-07-21","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.9,65.3846,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-07-27","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.9,65.3846,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-08-01","claude-opus-4-5","Claude Opus 4.5","Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.9,65.3846,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-07-21","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.3,78.8854,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-07-27","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.3,78.8854,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-08-01","claude-opus-4-5-thinking","Claude Opus 4.5 Thinking","Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.3,78.8854,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaomniscienceindex-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.5,79.0424,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaomniscienceindex-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.5,79.0424,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-thinking-aaomniscienceindex-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-6-max","Exact BenchLM registry variant Claude Opus 4.6 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",13.5,79.0424,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.5,71.1931,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.5,71.1931,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.5,71.1931,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaomniscienceindex-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.2,89.011,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaomniscienceindex-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.2,89.011,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-aaomniscienceindex-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.2,89.011,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.2,79.5918,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.2,79.5918,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.2,79.5918,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",27.4,89.9529,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",27.4,89.9529,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",27.4,89.9529,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaomniscienceindex-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",31.3,93.0141,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-aaomniscienceindex-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",31.3,93.0141,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-07-21","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.9,66.1695,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-07-27","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.9,66.1695,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-08-01","claude-sonnet-4-6","Claude Sonnet 4.6","Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.9,66.1695,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.3,80.4553,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.3,80.4553,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.3,80.4553,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaomniscienceindex-2026-07-21","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-4,65.3061,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaomniscienceindex-2026-07-27","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-4,65.3061,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-command-a-plus-aaomniscienceindex-2026-08-01","command-a-plus","Command A+","Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-4,65.3061,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaomniscienceindex-2026-07-21","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.3,36.0283,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaomniscienceindex-2026-07-27","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.3,36.0283,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-aaomniscienceindex-2026-08-01","deepseek-v3","DeepSeek V3","Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.3,36.0283,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaomniscienceindex-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaomniscienceindex-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-reasoning-aaomniscienceindex-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","deepseek-v3-1-reasoning-default","Exact BenchLM registry variant DeepSeek V3.1 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-07-21","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.1,36.1852,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-07-27","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.1,36.1852,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-08-01","deepseek-v3-1","DeepSeek V3.1","Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.1,36.1852,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-07-21","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.7,31.7896,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-07-27","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.7,31.7896,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-08-01","deepseek-v3-2","DeepSeek V3.2","Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.7,31.7896,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-aaomniscienceindex-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.3,50.9419,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-22.9,50.471,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-aaomniscienceindex-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-9.7,60.832,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10,60.5965,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaomniscienceindex-2026-07-21","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.1,47.1743,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaomniscienceindex-2026-07-27","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.1,47.1743,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-r1-aaomniscienceindex-2026-08-01","deepseek-r1","DeepSeek-R1","Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.1,47.1743,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-07-21","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-82.6,3.6107,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-07-27","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-82.6,3.6107,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-08-01","exaone-4-0-1-2b","Exaone 4.0 1.2B","Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-82.6,3.6107,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-07-21","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.3,19.5447,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-07-27","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.3,19.5447,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-08-01","exaone-4-0-32b","Exaone 4.0 32B","Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.3,19.5447,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-07-21","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-07-27","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-08-01","gemini-2-5-flash","Gemini 2.5 Flash","Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-07-21","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-14.3,57.2214,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-07-27","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-14.3,57.2214,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-08-01","gemini-2-5-pro","Gemini 2.5 Pro","Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-14.3,57.2214,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-07-21","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.6,65.6201,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-07-27","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.6,65.6201,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-08-01","gemini-3-flash","Gemini 3 Flash","Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-3.6,65.6201,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-07-21","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.8,80.8477,"index","higher","1.5.0","excluded","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-07-27","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.8,80.8477,"index","higher","1.6.0","excluded","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-08-01","gemini-3-pro","Gemini 3 Pro Preview","Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",15.8,80.8477,"index","higher","1.8.0","excluded","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-07-21","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.5,56.2794,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-07-27","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.5,56.2794,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-08-01","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.5,56.2794,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",32.9,94.27,"index","higher","1.5.0","excluded","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",32.9,94.27,"index","higher","1.6.0","excluded","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",32.9,94.27,"index","higher","1.8.0","excluded","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",22.7,86.2637,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",22.7,86.2637,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",22.7,86.2637,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.9,73.8619,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.9,73.8619,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.9,73.8619,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-07-21","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",23.5,86.8917,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-07-27","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",23.5,86.8917,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-08-01","gemini-3-6-flash","Gemini 3.6 Flash","Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",23.5,86.8917,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-07-21","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.9,16.719,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-07-27","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.9,16.719,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-08-01","gemma-3-27b","Gemma 3 27B","Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.9,16.719,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.9,27.708,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.9,27.708,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.9,27.708,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-07-21","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.1,30.6907,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-07-27","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.1,30.6907,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-08-01","gemma-4-26b-a4b","Gemma 4 26B A4B","Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.1,30.6907,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-07-21","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.4,32.81,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-07-27","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.4,32.81,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-08-01","gemma-4-31b","Gemma 4 31B","Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.4,32.81,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-07-21","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-24,49.6075,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-07-27","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-24,49.6075,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-08-01","gemma-4-e2b","Gemma 4 E2B","Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-24,49.6075,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-07-21","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-20,52.7473,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-07-27","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-20,52.7473,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-08-01","gemma-4-e4b","Gemma 4 E4B","Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-20,52.7473,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-07-21","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.5,19.3878,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-07-27","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.5,19.3878,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-08-01","glm-4-5-air","GLM-4.5-Air","Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-62.5,19.3878,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaomniscienceindex-2026-07-21","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.6,43.6421,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaomniscienceindex-2026-07-27","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.6,43.6421,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-6-aaomniscienceindex-2026-08-01","glm-4-6","GLM-4.6","Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.6,43.6421,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaomniscienceindex-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34.6,41.2873,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaomniscienceindex-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34.6,41.2873,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-aaomniscienceindex-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34.6,41.2873,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaomniscienceindex-2026-07-21","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2,70.0157,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaomniscienceindex-2026-07-27","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2,70.0157,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-aaomniscienceindex-2026-08-01","glm-5","GLM-5","Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2,70.0157,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-07-21","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.1,56.5934,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-07-27","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.1,56.5934,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-08-01","glm-5-turbo","GLM-5-Turbo","Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.1,56.5934,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaomniscienceindex-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.9,69.9372,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaomniscienceindex-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.9,69.9372,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-aaomniscienceindex-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.9,69.9372,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaomniscienceindex-2026-07-21","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4,71.5856,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaomniscienceindex-2026-07-27","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4,71.5856,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-2-aaomniscienceindex-2026-08-01","glm-5-2","GLM-5.2","Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4,71.5856,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-07-21","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19,53.5322,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-07-27","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19,53.5322,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-08-01","glm-5v-turbo","GLM-5V-Turbo","Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19,53.5322,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaomniscienceindex-2026-07-21","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.2,40.0314,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaomniscienceindex-2026-07-27","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.2,40.0314,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-aaomniscienceindex-2026-08-01","gpt-4-1","GPT-4.1","Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.2,40.0314,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-07-21","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.1,29.1209,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-07-27","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.1,29.1209,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-08-01","gpt-4-1-mini","GPT-4.1 mini","Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.1,29.1209,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-07-21","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.4,24.1758,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-07-27","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.4,24.1758,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-08-01","gpt-4-1-nano","GPT-4.1 nano","Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.4,24.1758,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaomniscienceindex-2026-07-21","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaomniscienceindex-2026-07-27","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-4o-aaomniscienceindex-2026-08-01","gpt-4o","GPT-4o","Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-21--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-27--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-08-01--configuration--gpt-5-high","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","gpt-5-high","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-21--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-27--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-08-01--configuration--gpt-5-medium","gpt-5","GPT-5","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","gpt-5-medium","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-21","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-27","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-high-aaomniscienceindex-2026-08-01","gpt-5-high","GPT-5 (high)","Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-21","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-27","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-08-01","gpt-5-medium","GPT-5 (medium)","Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.1,60.5181,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaomniscienceindex-2026-07-21","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.6,72.8414,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaomniscienceindex-2026-07-27","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.6,72.8414,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-aaomniscienceindex-2026-08-01","gpt-5-1","GPT-5.1","Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.6,72.8414,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-07-21","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-07-27","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-08-01","gpt-5-1-codex","GPT-5.1-Codex","Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-07-21","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-07-27","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-08-01","gpt-5-1-codex-max","GPT-5.1-Codex-Max","Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-6,63.7363,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaomniscienceindex-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-1,67.6609,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaomniscienceindex-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-1,67.6609,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-aaomniscienceindex-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-1,67.6609,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-07-21","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.5,66.4835,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-07-27","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.5,66.4835,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-08-01","gpt-5-2-codex","GPT-5.2 Codex","Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-2.5,66.4835,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-07-21","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",9.9,76.2166,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-07-27","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",9.9,76.2166,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-08-01","gpt-5-3-codex","GPT-5.3-Codex","Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",9.9,76.2166,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaomniscienceindex-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.7,72.9199,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaomniscienceindex-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.7,72.9199,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-aaomniscienceindex-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",5.7,72.9199,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-07-21","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.7,53.7677,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-07-27","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.7,53.7677,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-08-01","gpt-5-4-mini","GPT-5.4 mini","Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.7,53.7677,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-07-21","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.5,45.2904,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-07-27","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.5,45.2904,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-08-01","gpt-5-4-nano","GPT-5.4 nano","Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.5,45.2904,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaomniscienceindex-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",20.1,84.2229,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaomniscienceindex-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",20.1,84.2229,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-aaomniscienceindex-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",20.1,84.2229,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-11.2,59.6546,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-11.2,59.6546,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-11.2,59.6546,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",21.7,85.4788,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",21.7,85.4788,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",21.7,85.4788,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.2,68.2889,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.2,68.2889,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.2,68.2889,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-07-21","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50,29.1994,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-07-27","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50,29.1994,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-08-01","gpt-oss-120b","GPT-OSS 120B","Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50,29.1994,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-07-21","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-63.9,18.2889,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-07-27","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-63.9,18.2889,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-08-01","gpt-oss-20b","GPT-OSS 20B","Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-63.9,18.2889,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-07-21","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-81.8,4.2386,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-07-27","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-81.8,4.2386,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-08-01","granite-4-0-1b","Granite-4.0-1B","Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-81.8,4.2386,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-07-21","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72.1,11.8524,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-07-27","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72.1,11.8524,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-08-01","granite-4-0-350m","Granite-4.0-350M","Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72.1,11.8524,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-07-21","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-73.6,10.675,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-07-27","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-73.6,10.675,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-08-01","granite-4-0-h-1b","Granite-4.0-H-1B","Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-73.6,10.675,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-07-21","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-87.2,0,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-07-27","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-87.2,0,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-08-01","granite-4-0-h-350m","Granite-4.0-H-350M","Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-87.2,0,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaomniscienceindex-2026-07-21","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.8,71.4286,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaomniscienceindex-2026-07-27","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.8,71.4286,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-aaomniscienceindex-2026-08-01","grok-4","Grok 4","Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.8,71.4286,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-07-21","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-07-27","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-08-01","grok-4-fast-reasoning","Grok 4 Fast (Reasoning)","Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.4,46.1538,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-07-21","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.9,28.4929,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-07-27","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.9,28.4929,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-08-01","grok-4-1-fast","Grok 4.1 Fast","Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-50.9,28.4929,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-07-21","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.7,45.9184,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-07-27","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.7,45.9184,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-08-01","grok-4-1-fast-reasoning","Grok 4.1 Fast (Reasoning)","Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-28.7,45.9184,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaomniscienceindex-2026-07-21","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.3,82.81,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaomniscienceindex-2026-07-27","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.3,82.81,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-3-aaomniscienceindex-2026-08-01","grok-4-3","Grok 4.3","Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.3,82.81,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaomniscienceindex-2026-07-21","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.4,89.168,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaomniscienceindex-2026-07-27","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.4,89.168,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-5-aaomniscienceindex-2026-08-01","grok-4-5","Grok 4.5","Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",26.4,89.168,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-07-21","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36,40.1884,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-07-27","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36,40.1884,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-08-01","grok-code-fast-1","Grok Code Fast 1","Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36,40.1884,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaomniscienceindex-2026-07-21","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaomniscienceindex-2026-07-27","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-aaomniscienceindex-2026-08-01","hy3","Hy3","Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaomniscienceindex-2026-07-21","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaomniscienceindex-2026-07-27","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-hy3-preview-aaomniscienceindex-2026-08-01","hy3-preview","Hy3 Preview","Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-18.5,53.9246,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaomniscienceindex-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.1,70.0942,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaomniscienceindex-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.1,70.0942,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-aaomniscienceindex-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.1,70.0942,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaomniscienceindex-2026-07-21","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-57.9,22.9984,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaomniscienceindex-2026-07-27","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-57.9,22.9984,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-k-exaone-aaomniscienceindex-2026-08-01","k-exaone","K-Exaone","Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-57.9,22.9984,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaomniscienceindex-2026-07-21","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.5,46.8603,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaomniscienceindex-2026-07-27","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.5,46.8603,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-aaomniscienceindex-2026-08-01","kimi-k2","Kimi K2","Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-27.5,46.8603,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaomniscienceindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaomniscienceindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-aaomniscienceindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-8.1,62.0879,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaomniscienceindex-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.4,73.4694,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaomniscienceindex-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.4,73.4694,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-aaomniscienceindex-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",6.4,73.4694,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-07-21","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-07-27","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-08-01","kimi-k2-7-code","Kimi K2.7 Code","Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.7,60.0471,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaomniscienceindex-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.4,82.8885,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaomniscienceindex-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.4,82.8885,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-aaomniscienceindex-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18.4,82.8885,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-07-21","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-33.3,42.3077,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-07-27","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-33.3,42.3077,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-08-01","lfm2-5-8b-a1b","LFM2.5-8B-A1B","Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-33.3,42.3077,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-07-21","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-83.9,2.5903,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-07-27","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-83.9,2.5903,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-08-01","lfm2-5-vl-1-6b-extract","LFM2.5-VL-1.6B-Extract","Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-83.9,2.5903,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-07-21","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.7,16.876,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-07-27","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.7,16.876,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-08-01","ling-2-6-flash","Ling 2.6 Flash","Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-65.7,16.876,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-07-21","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.3,54.8666,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-07-27","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.3,54.8666,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-08-01","llama-3-1-405b","Llama 3.1 405B","Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.3,54.8666,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-07-21","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.8,35.6358,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-07-27","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.8,35.6358,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-08-01","llama-4-maverick","Llama 4 Maverick","Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-41.8,35.6358,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaomniscienceindex-2026-07-21","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-52.4,27.3155,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaomniscienceindex-2026-07-27","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-52.4,27.3155,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-llama-4-scout-aaomniscienceindex-2026-08-01","llama-4-scout","Llama 4 Scout","Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-52.4,27.3155,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-07-21","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.5,30.3768,"index","higher","1.5.0","excluded","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-07-27","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.5,30.3768,"index","higher","1.6.0","excluded","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-08-01","mimo-v2-flash","MiMo-V2-Flash","Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-48.5,30.3768,"index","higher","1.8.0","excluded","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-07-21","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.4,54.7881,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-07-27","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.4,54.7881,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-08-01","mimo-v2-omni","MiMo-V2-Omni","Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-17.4,54.7881,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-07-21","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.9,72.292,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-07-27","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.9,72.292,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-08-01","mimo-v2-pro","MiMo-V2-Pro","Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.9,72.292,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-07-21","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.6,71.2716,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-07-27","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.6,71.2716,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-08-01","mimo-v2-5-pro","MiMo-V2.5-Pro","Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",3.6,71.2716,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-07-21","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",0.7,68.9953,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-07-27","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",0.7,68.9953,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-08-01","minimax-m2-7","MiniMax M2.7","Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",0.7,68.9953,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaomniscienceindex-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.4,69.5447,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaomniscienceindex-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.4,69.5447,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-aaomniscienceindex-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",1.4,69.5447,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaomniscienceindex-2026-07-21","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34,41.7582,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaomniscienceindex-2026-07-27","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34,41.7582,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-2-aaomniscienceindex-2026-08-01","mistral-large-2","Mistral Large 2","Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-34,41.7582,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaomniscienceindex-2026-07-21","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.4,37.5196,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaomniscienceindex-2026-07-27","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.4,37.5196,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-large-3-aaomniscienceindex-2026-08-01","mistral-large-3","Mistral Large 3","Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.4,37.5196,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-07-21","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.5,43.7206,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-07-27","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.5,43.7206,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-08-01","mistral-medium-3","Mistral Medium 3","Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-31.5,43.7206,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-07-21","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.3,39.9529,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-07-27","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.3,39.9529,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-08-01","mistral-medium-3-5-128b","Mistral Medium 3.5 128B","Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-36.3,39.9529,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaomniscienceindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaomniscienceindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-reasoning-aaomniscienceindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","mistral-small-4-reasoning","Exact BenchLM registry variant Mistral Small 4 (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaomniscienceindex-2026-07-21","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaomniscienceindex-2026-07-27","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-mistral-small-4-aaomniscienceindex-2026-08-01","mistral-small-4","Mistral Small 4","Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.9,44.9765,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaomniscienceindex-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.1,71.6641,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaomniscienceindex-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.1,71.6641,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-aaomniscienceindex-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",4.1,71.6641,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18,82.5746,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18,82.5746,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",18,82.5746,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-07-21","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.6,27.9435,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-07-27","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.6,27.9435,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-08-01","nemotron-3-nano-30b","Nemotron 3 Nano 30B","Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-51.6,27.9435,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-07-21","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56,24.4898,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-07-27","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56,24.4898,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-08-01","nemotron-3-nano-omni-30b-a3b","Nemotron 3 Nano Omni 30B A3B","Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56,24.4898,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.8,67.8179,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.8,67.8179,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-0.8,67.8179,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-07-21","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.5,32.7316,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-07-27","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.5,32.7316,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-08-01","nemotron-ultra-253b","Nemotron Ultra 253B","Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-45.5,32.7316,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaomniscienceindex-2026-07-21","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaomniscienceindex-2026-07-27","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nova-pro-aaomniscienceindex-2026-08-01","nova-pro","Nova Pro","Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-47.6,31.0832,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaomniscienceindex-2026-07-21","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.5,60.2041,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaomniscienceindex-2026-07-27","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.5,60.2041,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o1-aaomniscienceindex-2026-08-01","o1","o1","Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-10.5,60.2041,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaomniscienceindex-2026-07-21","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.3,56.4364,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaomniscienceindex-2026-07-27","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.3,56.4364,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-o3-aaomniscienceindex-2026-08-01","o3","o3","Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-15.3,56.4364,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaomniscienceindex-2026-07-21","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.7,23.9403,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaomniscienceindex-2026-07-27","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.7,23.9403,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-phi-4-aaomniscienceindex-2026-08-01","phi-4","Phi-4","Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-56.7,23.9403,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-07-21","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",10.2,76.4521,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-07-27","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",10.2,76.4521,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-08-01","qwen3-6-max-preview","Qwen 3.6 Max (preview)","Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",10.2,76.4521,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaomniscienceindex-2026-07-21","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-43.1,34.6154,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaomniscienceindex-2026-07-27","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-43.1,34.6154,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-max-aaomniscienceindex-2026-08-01","qwen3-max","Qwen3 Max","Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-43.1,34.6154,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-42,35.4788,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaomniscienceindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaomniscienceindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-reasoning-aaomniscienceindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","qwen3-5-397b-thinking","Exact BenchLM registry variant Qwen3.5 397B (Reasoning); bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-29.8,45.0549,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.6,37.3626,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.6,37.3626,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-39.6,37.3626,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.4,32.0251,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.4,32.0251,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-46.4,32.0251,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-07-21","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19.8,52.9042,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-07-27","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19.8,52.9042,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-08-01","qwen3-6-27b","Qwen3.6 27B","Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-19.8,52.9042,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-07-21","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.7,70.5651,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-07-27","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.7,70.5651,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-08-01","qwen3-6-plus","Qwen3.6 Plus","Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.7,70.5651,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-07-21","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-21.4,51.6484,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-07-27","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-21.4,51.6484,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-08-01","qwen3-6-35b-a3b","Qwen3.6-35B-A3B","Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-21.4,51.6484,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.1,79.5133,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.1,79.5133,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",14.1,79.5133,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.4,70.3297,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.4,70.3297,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",2.4,70.3297,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaomniscienceindex-2026-07-21","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-59.5,21.7425,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaomniscienceindex-2026-07-27","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-59.5,21.7425,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-105b-aaomniscienceindex-2026-08-01","sarvam-105b","Sarvam 105B","Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-59.5,21.7425,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaomniscienceindex-2026-07-21","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72,11.9309,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaomniscienceindex-2026-07-27","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72,11.9309,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sarvam-30b-aaomniscienceindex-2026-08-01","sarvam-30b","Sarvam 30B","Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-72,11.9309,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaomniscienceindex-2026-07-21","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-61.7,20.0157,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaomniscienceindex-2026-07-27","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-61.7,20.0157,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-solar-pro-2-aaomniscienceindex-2026-08-01","solar-pro-2","Solar Pro 2","Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-61.7,20.0157,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-37.5,39.011,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-37.5,39.011,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-37.5,39.011,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-07-21","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-07-27","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-08-01","trinity-large-preview","Trinity-Large-Preview","Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-07-21","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.5.0","reference-only","composite","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-07-27","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.6.0","reference-only","composite","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-08-01","trinity-large-thinking","Trinity-Large-Thinking","Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","aa-omniscience","AA-Omniscience","research","Artificial Analysis","2026",-44.2,33.752,"index","higher","1.8.0","reference-only","composite","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-906","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)",null,"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,40.15,40.15,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-906--configuration--claude-fable-5-max","claude-fable-5","Claude Fable 5","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","claude-fable-5-max","Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,40.15,40.15,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1189","claude-haiku-4-5","Claude Haiku 4.5","Claude 4.5 Haiku (Reasoning)",null,"Claude 4.5 Haiku (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-4.2167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-986","claude-opus-4-5","Claude Opus 4.5","Claude Opus 4.5 (Reasoning)",null,"Claude Opus 4.5 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,13.2667,13.2667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1033","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,13.5,13.5,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1033--configuration--claude-opus-4-6-max","claude-opus-4-6","Claude Opus 4.6","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","claude-opus-4-6-max","Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,13.5,13.5,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1020","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,26.1667,26.1667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1020--configuration--claude-opus-4-7-max","claude-opus-4-7","Claude Opus 4.7","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","claude-opus-4-7-max","Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,26.1667,26.1667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1172","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)",null,"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,27.4333,27.4333,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1172--configuration--claude-opus-4-8-max","claude-opus-4-8","Claude Opus 4.8","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","claude-opus-4-8-max","Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,27.4333,27.4333,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1770","claude-sonnet-3-7","Claude Sonnet 3.7","Claude 3.7 Sonnet (Reasoning)",null,"Claude 3.7 Sonnet (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,0.3167,0.3167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-3-7-sonnet-thinking."],["evidence-2026-07-1718","claude-sonnet-4","Claude Sonnet 4","Claude 4 Sonnet (Reasoning)",null,"Claude 4 Sonnet (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,0.1833,0.1833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for claude-4-sonnet-thinking."],["evidence-2026-07-1372","claude-sonnet-4-5","Claude Sonnet 4.5","Claude 4.5 Sonnet (Reasoning)",null,"Claude 4.5 Sonnet (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,0.3333,0.3333,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for claude-4-5-sonnet-thinking."],["evidence-2026-07-1010","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,12.3667,12.3667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1010--configuration--claude-sonnet-4-6-max","claude-sonnet-4-6","Claude Sonnet 4.6","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","claude-sonnet-4-6-max","Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,12.3667,12.3667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-958","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)",null,"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,15.3167,15.3167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-958--configuration--claude-sonnet-5-max","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","claude-sonnet-5-max","Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,15.3167,15.3167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1626","command-a-plus","Command A+","Command A+",null,"Command A+","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-3.9833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for command-a-plus."],["evidence-2026-07-1532","deepseek-v3-1-terminus","DeepSeek V3.1 Terminus","DeepSeek V3.1 Terminus (Reasoning)",null,"DeepSeek V3.1 Terminus (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-24.3167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-1-terminus-reasoning."],["evidence-2026-07-1517","deepseek-v3-2","DeepSeek V3.2","DeepSeek V3.2 (Reasoning)",null,"DeepSeek V3.2 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-20.8833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for deepseek-v3-2-reasoning."],["evidence-2026-07-1156","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)",null,"DeepSeek V4 Flash (Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-22.9,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1156--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash (Reasoning, Max Effort)","deepseek-v4-flash-max","DeepSeek V4 Flash (Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-22.9,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1099","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)",null,"DeepSeek V4 Pro (Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.0167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1099--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","DeepSeek V4 Pro (Reasoning, Max Effort)","deepseek-v4-pro-max","DeepSeek V4 Pro (Reasoning, Max Effort)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.0167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1832","gemini-2-5-flash","Gemini 2.5 Flash","Gemini 2.5 Flash Preview (Sep '25) (Reasoning)",null,"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-34.7167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-flash-preview-09-2025-reasoning."],["evidence-2026-07-1792","gemini-2-5-pro","Gemini 2.5 Pro","Gemini 2.5 Pro",null,"Gemini 2.5 Pro","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-14.3,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemini-2-5-pro."],["evidence-2026-07-1130","gemini-3-flash","Gemini 3 Flash","Gemini 3 Flash Preview (Reasoning)",null,"Gemini 3 Flash Preview (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,11.5667,11.5667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1208","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)",null,"Gemini 3 Pro Preview (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,15.8,15.8,"index","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","excluded","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1208--configuration--gemini-3-pro-high","gemini-3-pro","Gemini 3 Pro Preview","Gemini 3 Pro Preview (high)","gemini-3-pro-high","Gemini 3 Pro Preview (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,15.8,15.8,"index","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","excluded","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1071","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","Gemini 3.1 Flash-Lite",null,"Gemini 3.1 Flash-Lite","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-15.5167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1214","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Gemini 3.1 Pro Preview",null,"Gemini 3.1 Pro Preview","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,32.9333,32.9333,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-915","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)",null,"Gemini 3.5 Flash (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,22.6833,22.6833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-915--configuration--gemini-3-5-flash-high","gemini-3-5-flash","Gemini 3.5 Flash","Gemini 3.5 Flash (high)","gemini-3-5-flash-high","Gemini 3.5 Flash (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,22.6833,22.6833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-2046","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Artificial Analysis Omniscience Index independent evaluation.",null,"Artificial Analysis Omniscience Index independent evaluation.","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,6.9,6.9,"index","higher","1.4.1","reference-only","composite","2026-07-21","2026-07-21","production::aa-gemini-3-5-flash-lite","aa-gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-5-flash-lite","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-Omniscience Index 6.9."],["evidence-2026-07-2006","gemini-3-6-flash","Gemini 3.6 Flash","Artificial Analysis Omniscience Index independent evaluation.",null,"Artificial Analysis Omniscience Index independent evaluation.","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,23.5,23.5,"index","higher","1.4.1","reference-only","composite","2026-07-21","2026-07-21","production::aa-gemini-3-6-flash","aa-gemini-3-6-flash","Gemini 3.6 Flash (Artificial Analysis)","Artificial Analysis","https://artificialanalysis.ai/models/gemini-3-6-flash","2026-07-21","2026-07-21","2026-07-21","source-checked","AA-Omniscience Index 23.5."],["evidence-2026-07-1876","gemma-4-12b","Gemma 4 12B Unified","Gemma 4 12B (Reasoning)",null,"Gemma 4 12B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-51.8833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for gemma-4-12b."],["evidence-2026-07-1612","gemma-4-26b","Gemma 4 26B","Gemma 4 26B A4B (Reasoning)",null,"Gemma 4 26B A4B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-48.0667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gemma-4-26b-a4b."],["evidence-2026-07-1225","gemma-4-31b","Gemma 4 31B","Gemma 4 31B (Reasoning)",null,"Gemma 4 31B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-45.4167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1733","glm-4-6","GLM-4.6","GLM-4.6 (Reasoning)",null,"GLM-4.6 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-41.7333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for glm-4-6-reasoning."],["evidence-2026-07-1546","glm-4-7","GLM-4.7","GLM-4.7 (Reasoning)",null,"GLM-4.7 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-34.6,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for glm-4-7."],["evidence-2026-07-1027","glm-5","GLM-5","GLM-5 (Reasoning)",null,"GLM-5 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,2,2,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-964","glm-5-turbo","GLM-5-Turbo","GLM-5-Turbo",null,"GLM-5-Turbo","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-15.0833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1086","glm-5-1","GLM-5.1","GLM-5.1 (Reasoning)",null,"GLM-5.1 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,1.9333,1.9333,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1251","glm-5-2","GLM-5.2","GLM-5.2 (max)",null,"GLM-5.2 (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,3.9667,3.9667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1251--configuration--glm-5-2-max","glm-5-2","GLM-5.2","GLM-5.2 (max)","glm-5-2-max","GLM-5.2 (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,3.9667,3.9667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-981","glm-5v-turbo","GLM-5V-Turbo","GLM 5V Turbo (Reasoning)",null,"GLM 5V Turbo (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-18.9833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1332","gpt-5","GPT-5","GPT-5 (high)",null,"GPT-5 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-8.0833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1332--configuration--gpt-5-high","gpt-5","GPT-5","GPT-5 (high)","gpt-5-high","GPT-5 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-8.0833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5."],["evidence-2026-07-1319","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)",null,"GPT-5 Codex (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-6.7667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1319--configuration--gpt-5-codex-high","gpt-5-codex","GPT-5 Codex","GPT-5 Codex (high)","gpt-5-codex-high","GPT-5 Codex (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-6.7667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-codex."],["evidence-2026-07-1346","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)",null,"GPT-5 mini (medium)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.7833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1346--configuration--gpt-5-mini-medium","gpt-5-mini","GPT-5 mini","GPT-5 mini (medium)","gpt-5-mini-medium","GPT-5 mini (medium)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.7833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-mini-medium."],["evidence-2026-07-1064","gpt-5-1","GPT-5.1","GPT-5.1 (high)",null,"GPT-5.1 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,5.55,5.55,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1064--configuration--gpt-5-1-high","gpt-5-1","GPT-5.1","GPT-5.1 (high)","gpt-5-1-high","GPT-5.1 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,5.55,5.55,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1298","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)",null,"GPT-5.2 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-1,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1298--configuration--gpt-5-2-xhigh","gpt-5-2","GPT-5.2","GPT-5.2 (xhigh)","gpt-5-2-xhigh","GPT-5.2 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-1,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2."],["evidence-2026-07-1309","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)",null,"GPT-5.2 Codex (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-2.4833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1309--configuration--gpt-5-2-codex-xhigh","gpt-5-2-codex","GPT-5.2 Codex","GPT-5.2 Codex (xhigh)","gpt-5-2-codex-xhigh","GPT-5.2 Codex (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-2.4833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-2-codex."],["evidence-2026-07-1080","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)",null,"GPT-5.3 Codex (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,9.8833,9.8833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1080--configuration--gpt-5-3-codex-xhigh","gpt-5-3-codex","GPT-5.3-Codex","GPT-5.3 Codex (xhigh)","gpt-5-3-codex-xhigh","GPT-5.3 Codex (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,9.8833,9.8833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1047","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)",null,"GPT-5.4 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,5.65,5.65,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1047--configuration--gpt-5-4-xhigh","gpt-5-4","GPT-5.4","GPT-5.4 (xhigh)","gpt-5-4-xhigh","GPT-5.4 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,5.65,5.65,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1282","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)",null,"GPT-5.4 mini (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-18.6833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1282--configuration--gpt-5-4-mini-xhigh","gpt-5-4-mini","GPT-5.4 mini","GPT-5.4 mini (xhigh)","gpt-5-4-mini-xhigh","GPT-5.4 mini (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-18.6833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for gpt-5-4-mini."],["evidence-2026-07-1145","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)",null,"GPT-5.4 nano (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-29.55,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1145--configuration--gpt-5-4-nano-xhigh","gpt-5-4-nano","GPT-5.4 nano","GPT-5.4 nano (xhigh)","gpt-5-4-nano-xhigh","GPT-5.4 nano (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-29.55,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-970","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)",null,"GPT-5.5 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,20.0667,20.0667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-970--configuration--gpt-5-5-xhigh","gpt-5-5","GPT-5.5","GPT-5.5 (xhigh)","gpt-5-5-xhigh","GPT-5.5 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,20.0667,20.0667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-936","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)",null,"GPT-5.6 Luna (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-11.2333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-936--configuration--gpt-5-6-luna-max","gpt-5-6-luna","GPT-5.6 Luna","GPT-5.6 Luna (max)","gpt-5-6-luna-max","GPT-5.6 Luna (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-11.2333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-943","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)",null,"GPT-5.6 Sol (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,21.7,21.7,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-943--configuration--gpt-5-6-sol-max","gpt-5-6-sol","GPT-5.6 Sol","GPT-5.6 Sol (max)","gpt-5-6-sol-max","GPT-5.6 Sol (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,21.7,21.7,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-992","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)",null,"GPT-5.6 Terra (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-0.2167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-992--configuration--gpt-5-6-terra-max","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra (max)","gpt-5-6-terra-max","GPT-5.6 Terra (max)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-0.2167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1865","grok-3-mini","Grok 3 mini","Grok 3 mini Reasoning (high)",null,"Grok 3 mini Reasoning (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-6.05,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-3-mini-reasoning."],["evidence-2026-07-1653","grok-4","Grok 4","Grok 4",null,"Grok 4","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,3.75,3.75,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4."],["evidence-2026-07-1757","grok-4-fast","Grok 4 Fast","Grok 4 Fast (Reasoning)",null,"Grok 4 Fast (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-28.4,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-4-fast-reasoning."],["evidence-2026-07-1442","grok-4-20","Grok 4.20","Grok 4.20 0309 v2 (Reasoning)",null,"Grok 4.20 0309 v2 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,15.35,15.35,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for grok-4-20."],["evidence-2026-07-1110","grok-4-3","Grok 4.3","Grok 4.3 (high)",null,"Grok 4.3 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,18.3167,18.3167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1138","grok-4-5","Grok 4.5","Grok 4.5 (high)",null,"Grok 4.5 (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,26.3833,26.3833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1887","grok-code-fast-1","Grok Code Fast 1","Grok Code Fast 1",null,"Grok Code Fast 1","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-36,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for grok-code-fast-1."],["evidence-2026-07-1844","kimi-k2-0905","Kimi K2 0905","Kimi K2 0905",null,"Kimi K2 0905","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-26.4667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-0905."],["evidence-2026-07-1665","kimi-k2-thinking","Kimi K2 Thinking","Kimi K2 Thinking",null,"Kimi K2 Thinking","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-20.4833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for kimi-k2-thinking."],["evidence-2026-07-1181","kimi-k2-5","Kimi K2.5","Kimi K2.5 (Reasoning)",null,"Kimi K2.5 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-8.1167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1410","kimi-k2-6","Kimi K2.6","Kimi K2.6",null,"Kimi K2.6","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,6.4167,6.4167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-6."],["evidence-2026-07-1428","kimi-k2-7-code","Kimi K2.7 Code","Kimi K2.7 Code",null,"Kimi K2.7 Code","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.7,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for kimi-k2-7-code."],["evidence-2026-07-1949","kimi-k3","Kimi K3","Kimi K3; Artificial Analysis Omniscience Index.",null,"Kimi K3; Artificial Analysis Omniscience Index.","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,18.4,18.4,"index","higher","1.4.1","reference-only","composite","2026-07-20","2026-07-20","production::aa-model-kimi-k3","aa-model-kimi-k3","Kimi K3 Intelligence, Performance & Price Analysis","Artificial Analysis","https://artificialanalysis.ai/models/kimi-k3","2026-07-20","2026-07-21","2026-07-21","source-checked","AA-Omniscience index 18.4; accuracy 46.0 and hallucination rate 50.9 also published. Negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1782","ling-2-6-1t","Ling 2.6 1T","Ling-2.6-1T",null,"Ling-2.6-1T","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-51,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for ling-2-6-1t."],["evidence-2026-07-1582","mimo-v2-flash","MiMo-V2-Flash","MiMo-V2-Flash (Reasoning)",null,"MiMo-V2-Flash (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-43.45,0,"index","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","excluded","Imported from Artificial Analysis model evaluation payload for mimo-v2-flash-reasoning."],["evidence-2026-07-1166","mimo-v2-omni","MiMo-V2-Omni","MiMo-V2-Omni",null,"MiMo-V2-Omni","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-17.4333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1093","mimo-v2-pro","MiMo-V2-Pro","MiMo-V2-Pro",null,"MiMo-V2-Pro","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,4.9167,4.9167,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1569","mimo-v2-5","MiMo-V2.5","MiMo-V2.5",null,"MiMo-V2.5","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-9.3333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for mimo-v2-5-0424."],["evidence-2026-07-926","mimo-v2-5-pro","MiMo-V2.5-Pro","MiMo-V2.5-Pro",null,"MiMo-V2.5-Pro","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,3.6,3.6,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1746","minimax-m2","MiniMax M2","MiniMax-M2",null,"MiniMax-M2","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-46.9333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2."],["evidence-2026-07-1685","minimax-m2-1","MiniMax M2.1","MiniMax-M2.1",null,"MiniMax-M2.1","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-32.8333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for minimax-m2-1."],["evidence-2026-07-1559","minimax-m2-5","MiniMax M2.5","MiniMax-M2.5",null,"MiniMax-M2.5","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-39.7,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for minimax-m2-5."],["evidence-2026-07-1055","minimax-m2-7","MiniMax M2.7","MiniMax-M2.7",null,"MiniMax-M2.7","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,0.6833,0.6833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1001","minimax-m3","MiniMax M3","MiniMax-M3",null,"MiniMax-M3","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,1.3667,1.3667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1040","mistral-large-3","Mistral Large 3","Mistral Large 3",null,"Mistral Large 3","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-39.4333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1241","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5",null,"Mistral Medium 3.5","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-36.3167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-951","mistral-small-4","Mistral Small 4","Mistral Small 4 (Reasoning)",null,"Mistral Small 4 (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-29.9,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1262","muse-spark","Muse Spark","Muse Spark",null,"Muse Spark","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,4.0833,4.0833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1235","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)",null,"Muse Spark 1.1 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,18,18,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1235--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh)","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,18,18,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1920","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.",null,"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,40.6,40.6,"percent","higher","1.4.1","reference-only","composite","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1920--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","muse-spark-1-1-xhigh","Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,40.6,40.6,"percent","higher","1.4.1","reference-only","composite","2026-07-16","2026-07-16","production::aa-model-muse-spark-1-1","aa-model-muse-spark-1-1","Muse Spark 1.1 (xhigh) analysis","Artificial Analysis","https://artificialanalysis.ai/models/muse-spark-1-1","2026-07-16","2026-07-16","2026-07-16","source-checked","Extracted from Artificial Analysis public model/evaluation pages as cited on BenchLM model profile 2026-07-16."],["evidence-2026-07-1819","nemotron-3-super","Nemotron 3 Super","NVIDIA Nemotron 3 Super 120B A12B (Reasoning)",null,"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-42.0667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for nvidia-nemotron-3-super-120b-a12b."],["evidence-2026-07-1597","nemotron-3-ultra","Nemotron 3 Ultra","Nemotron 3 Ultra 550B A55B (Reasoning)",null,"Nemotron 3 Ultra 550B A55B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-0.8333,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nvidia-nemotron-3-ultra-550b-a55b."],["evidence-2026-07-1639","nova-2-pro","Nova 2 Pro","Nova 2.0 Pro Preview (medium)",null,"Nova 2.0 Pro Preview (medium)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-48.05,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for nova-2-0-pro-reasoning-medium."],["nvidia-nemotron-3-5-lightning-aa-omniscience-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,17.5,17.5,"index","higher","1.8.0","reference-only","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1855","o1","o1","o1",null,"o1","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-10.55,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o1."],["evidence-2026-07-1359","o3","o3","o3",null,"o3","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-15.2667,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for o3."],["evidence-2026-07-1806","o4-mini","o4-mini","o4-mini (high)",null,"o4-mini (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-35.75,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1806--configuration--o4-mini-high","o4-mini","o4-mini","o4-mini (high)","o4-mini-high","o4-mini (high)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-35.75,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for o4-mini."],["evidence-2026-07-1676","qwen3-max","Qwen3 Max","Qwen3 Max Thinking",null,"Qwen3 Max Thinking","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-34.4167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-max-thinking."],["evidence-2026-07-1493","qwen3-5-122b","Qwen3.5 122B","Qwen3.5 122B A10B (Reasoning)",null,"Qwen3.5 122B A10B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-39.5833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-122b-a10b."],["evidence-2026-07-1505","qwen3-5-27b","Qwen3.5 27B","Qwen3.5 27B (Reasoning)",null,"Qwen3.5 27B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-42.0167,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-27b."],["evidence-2026-07-1706","qwen3-5-35b","Qwen3.5 35B","Qwen3.5 35B A3B (Reasoning)",null,"Qwen3.5 35B A3B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-46.3833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-35b-a3b."],["evidence-2026-07-1475","qwen3-5-397b","Qwen3.5 397B A17B","Qwen3.5 397B A17B (Reasoning)",null,"Qwen3.5 397B A17B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-29.7833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-5-397b-a17b."],["evidence-2026-07-1696","qwen3-5-omni-plus","Qwen3.5 Omni Plus","Qwen3.5 Omni Plus",null,"Qwen3.5 Omni Plus","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-12.3,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Wave 2 import from Artificial Analysis evaluation payload for qwen3-5-omni-plus."],["evidence-2026-07-1453","qwen3-6-27b","Qwen3.6 27B","Qwen3.6 27B (Reasoning)",null,"Qwen3.6 27B (Reasoning)","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,-19.7833,0,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-27b."],["evidence-2026-07-1465","qwen3-6-max","Qwen3.6 Max","Qwen3.6 Max Preview",null,"Qwen3.6 Max Preview","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,10.2,10.2,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from Artificial Analysis model evaluation payload for qwen3-6-max."],["evidence-2026-07-1269","qwen3-6-plus","Qwen3.6 Plus","Qwen3.6 Plus",null,"Qwen3.6 Plus","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,2.65,2.65,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1121","qwen-3-7-max","Qwen3.7-Max","Qwen3.7 Max",null,"Qwen3.7 Max","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,14.0833,14.0833,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-1199","qwen-3-7-plus","Qwen3.7-Plus","Qwen3.7 Plus",null,"Qwen3.7 Plus","aa-omniscience","AA-Omniscience","research","Artificial Analysis",null,2.3667,2.3667,"index","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::aa-omniscience-leaderboard","aa-omniscience-leaderboard","AA-Omniscience Leaderboard","Artificial Analysis","https://artificialanalysis.ai/evaluations/omniscience","2026-07-15","2026-07-15","2026-07-15","source-checked","AA-Omniscience index; negatives clamped to 0 on the normalized 0–100 scale."],["evidence-2026-07-395","claude-fable-5","Claude Fable 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",89,89,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-390","claude-mythos-5","Claude Mythos 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",87.8,87.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-523","claude-opus-4-5","Claude Opus 4.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",80.9,80.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-456","claude-opus-4-6","Claude Opus 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",91.5,91.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-405","claude-opus-4-8","Claude Opus 4.8","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",88.1,88.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-511","claude-sonnet-4-6","Claude Sonnet 4.6","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",83.6,83.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-496","claude-sonnet-5","Claude Sonnet 5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",87.3,87.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-555","deepseek-v4-pro","DeepSeek V4 Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",48.4,48.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-550","gemini-3-flash","Gemini 3 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",43.8,43.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-540","gemini-3-pro-deep-think","Gemini 3 Pro Deep Think","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",83.6,83.6,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-469","gemini-3-pro","Gemini 3 Pro Preview","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",77.1,77.1,"percent","higher","1.3.0","excluded","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","excluded","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-423","gemini-3-5-flash","Gemini 3.5 Flash","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",71.9,71.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-545","gemma-4-31b","Gemma 4 31B","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",65.2,65.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-491","glm-5","GLM-5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",82.7,82.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-462","glm-5-1","GLM-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",83.7,83.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-532","glm-5-2","GLM-5.2","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",84.5,84.5,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-517","gpt-5-1","GPT-5.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",76.9,76.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-486","gpt-5-3-codex","GPT-5.3-Codex","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",89,89,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-429","gpt-5-4","GPT-5.4","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",96.8,96.8,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-480","gpt-5-4-nano","GPT-5.4 nano","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",67.6,67.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-417","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",82.2,82.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-536","gpt-5-5","GPT-5.5","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","gpt-5-5-pro-default-high","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",87.1,87.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-474","gpt-5-6-luna","GPT-5.6 Luna","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",75.2,75.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-400","gpt-5-6-sol","GPT-5.6 Sol","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",77,77,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-434","gpt-5-6-terra","GPT-5.6 Terra","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",75.6,75.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-558","grok-4-1","Grok 4.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",90.1,90.1,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-500","grok-4-3","Grok 4.3","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",63.7,63.7,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-465","mimo-v2-pro","MiMo-V2-Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",71,71,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-448","mimo-v2-5-pro","MiMo-V2.5-Pro","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",77.4,77.4,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-528","minimax-m2-7","MiniMax M2.7","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",52.9,52.9,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-443","muse-spark","Muse Spark","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",79.2,79.2,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-410","muse-spark-1-1","Muse Spark 1.1","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",88.3,88.3,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-505","qwen3-6-plus","Qwen3.6 Plus","BenchLM public leaderboard / CursorBench snapshot for the exact model variant.",null,"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","bl-research","BenchLM Knowledge prior","research","BenchLM","bench-align-v5.1",67.6,67.6,"percent","higher","1.3.0","reference-only","composite","2026-07-15","2026-07-15","production::benchlm-leaderboard","benchlm-leaderboard","BenchLM overall leaderboard snapshot","BenchLM","https://benchlm.ai/","2026-07-15","2026-07-16","2026-07-15","unverified","BenchLM category prior (knowledge) used to estimate missing category coverage under methodology 1.3.0."],["evidence-2026-07-2068","claude-opus-5","Claude Opus 5","BioMysteryBench hard split as published in the Anthropic Claude Opus 5 launch table.",null,"BioMysteryBench hard split as published in the Anthropic Claude Opus 5 launch table.","biomystery-bench","BioMysteryBench","research","Anthropic","hard",49.4,49.4,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 49.4% on BioMysteryBench hard (90.1% human-solved subset also published)."],["evidence-2026-07-2069","claude-opus-5","Claude Opus 5","BioMysteryBench human-solved subset as published in the Anthropic Claude Opus 5 launch table.",null,"BioMysteryBench human-solved subset as published in the Anthropic Claude Opus 5 launch table.","biomystery-bench","BioMysteryBench","research","Anthropic","human-solved",90.1,90.1,"percent","higher","1.6.0","reference-only","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 90.1% on the BioMysteryBench human-solved subset."],["benchlm-ref-agents-a1-browsecomp-2026-07-21","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.51,65.0837,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-browsecomp-2026-07-27","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.51,65.0837,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-agents-a1-browsecomp-2026-08-01","agents-a1","Agents-A1","Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.51,65.0837,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-browsecomp-2026-07-21","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",88,91.2134,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-browsecomp-2026-07-27","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",88,91.2134,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-mythos-5-browsecomp-2026-08-01","claude-mythos-5","Claude Mythos 5","Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",88,91.2134,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-browsecomp-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.7,82.2176,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-browsecomp-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.7,82.2176,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-browsecomp-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.7,82.2176,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-browsecomp-2026-07-21","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",79.3,73.0126,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-browsecomp-2026-07-27","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",79.3,73.0126,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-7-adaptive-browsecomp-2026-08-01","claude-opus-4-7","Claude Opus 4.7","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","claude-opus-4-7-max","Exact BenchLM registry variant Claude Opus 4.7 (Adaptive); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",79.3,73.0126,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-browsecomp-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.3,83.4728,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-browsecomp-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.3,83.4728,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-browsecomp-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.3,83.4728,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-browsecomp-2026-07-27","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",90.8,97.0711,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-5-browsecomp-2026-08-01","claude-opus-5","Claude Opus 5","Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",90.8,97.0711,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-browsecomp-2026-07-21","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.7,84.3096,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-browsecomp-2026-07-27","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.7,84.3096,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-sonnet-5-browsecomp-2026-08-01","claude-sonnet-5","Claude Sonnet 5","Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.7,84.3096,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-07-21","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-07-21--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-07-27","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-07-27--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-08-01","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-high-browsecomp-2026-08-01--configuration--deepseek-v4-flash-high","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-high","Exact BenchLM registry variant DeepSeek V4 Flash (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",53.5,19.0377,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-21--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-27--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-08-01--configuration--deepseek-v4-flash-max","deepseek-v4-flash","DeepSeek V4 Flash","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-flash-max","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-21","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-27","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-08-01","deepseek-v4-flash-max","DeepSeek V4 Flash (Max)","Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",73.2,60.251,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-07-21","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-07-21--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-07-27","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-07-27--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-08-01","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-high-browsecomp-2026-08-01--configuration--deepseek-v4-pro-high","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-high","Exact BenchLM registry variant DeepSeek V4 Pro (High); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",80.4,75.3138,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-21--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-27--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-08-01--configuration--deepseek-v4-pro-max","deepseek-v4-pro","DeepSeek V4 Pro","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","deepseek-v4-pro-max","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-21","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-27","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-08-01","deepseek-v4-pro-max","DeepSeek V4 Pro (Max)","Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.4,81.59,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-browsecomp-2026-07-21","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",52,15.8996,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-browsecomp-2026-07-27","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",52,15.8996,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-4-7-browsecomp-2026-08-01","glm-4-7","GLM-4.7","Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",52,15.8996,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-browsecomp-2026-07-21","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",68,49.3724,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-browsecomp-2026-07-27","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",68,49.3724,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-glm-5-1-browsecomp-2026-08-01","glm-5-1","GLM-5.1","Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",68,49.3724,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-browsecomp-2026-07-21","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",65.8,44.7699,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-browsecomp-2026-07-27","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",65.8,44.7699,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-2-browsecomp-2026-08-01","gpt-5-2","GPT-5.2","Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",65.8,44.7699,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-browsecomp-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",89.3,93.9331,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-browsecomp-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",89.3,93.9331,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-pro-browsecomp-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-4-pro-default-medium","Exact BenchLM registry variant GPT-5.4 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",89.3,93.9331,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-browsecomp-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",82.7,80.1255,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-browsecomp-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",82.7,80.1255,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-browsecomp-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",82.7,80.1255,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-browsecomp-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",90.1,95.6067,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-browsecomp-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",90.1,95.6067,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-pro-browsecomp-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","gpt-5-5-pro-default-high","Exact BenchLM registry variant GPT-5.5 Pro; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",90.1,95.6067,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-browsecomp-2026-07-21","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.4,83.682,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-browsecomp-2026-07-27","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.4,83.682,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-5-browsecomp-2026-08-01","gpt-5-5","GPT-5.5","Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",84.4,83.682,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-browsecomp-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.3,81.3808,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-browsecomp-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.3,81.3808,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-browsecomp-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.3,81.3808,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-browsecomp-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",92.2,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-browsecomp-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",92.2,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-browsecomp-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",92.2,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-browsecomp-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",87.5,90.1674,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-browsecomp-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",87.5,90.1674,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-browsecomp-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",87.5,90.1674,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-browsecomp-2026-07-21","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",77.1,68.41,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-browsecomp-2026-07-27","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",77.1,68.41,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-browsecomp-2026-08-01","inkling","Inkling","Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",77.1,68.41,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-inkling-small-browsecomp-2026-08-01","inkling-small","Inkling-Small","Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",77.4,69.0377,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-browsecomp-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-browsecomp-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-reasoning-browsecomp-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","kimi-k2-5-thinking","Exact BenchLM registry variant Kimi K2.5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-browsecomp-2026-07-21","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-browsecomp-2026-07-27","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-k2-5-browsecomp-2026-08-01","kimi-k2-5","Kimi K2.5","Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",60.6,33.8912,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-browsecomp-2026-07-21","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.2,81.1715,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-browsecomp-2026-07-27","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.2,81.1715,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-2-6-browsecomp-2026-08-01","kimi-k2-6","Kimi K2.6","Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.2,81.1715,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-browsecomp-2026-07-21","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",91.2,97.9079,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-browsecomp-2026-07-27","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",91.2,97.9079,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-kimi-3-browsecomp-2026-08-01","kimi-k3","Kimi K3","Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",91.2,97.9079,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-browsecomp-2026-07-21","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.52,81.841,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-browsecomp-2026-07-27","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.52,81.841,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-minimax-m3-browsecomp-2026-08-01","minimax-m3","MiniMax M3","Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",83.52,81.841,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-browsecomp-2026-07-21","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",44.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-browsecomp-2026-07-27","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",44.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-nemotron-3-ultra-browsecomp-2026-08-01","nemotron-3-ultra","Nemotron 3 Ultra","Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",44.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-browsecomp-2026-07-21","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-browsecomp-2026-07-27","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-27b-browsecomp-2026-08-01","qwen3-5-27b","Qwen3.5 27B","Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-browsecomp-2026-07-21","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",62,36.8201,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-browsecomp-2026-07-27","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",62,36.8201,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-397b-browsecomp-2026-08-01","qwen3-5-397b","Qwen3.5 397B A17B","Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",62,36.8201,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-07-21","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",63.8,40.5858,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-07-27","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",63.8,40.5858,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-08-01","qwen3-5-122b-a10b","Qwen3.5-122B-A10B","Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",63.8,40.5858,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-07-21","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-07-27","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-08-01","qwen3-5-35b-a3b","Qwen3.5-35B-A3B","Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",61,34.728,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-browsecomp-2026-07-21","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.82,65.7322,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-browsecomp-2026-07-27","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.82,65.7322,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-step-3-7-flash-browsecomp-2026-08-01","step-3-7-flash","Step 3.7 Flash","Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","browsecomp","BrowseComp","research","OpenAI","2025",75.82,65.7322,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-08-15-longcat-2-0-browsecomp-standard","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","browsecomp","BrowseComp","research","OpenAI","standard",79.9,79.9,"percent","higher","2.1.0","ranking-eligible","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 owner score."],["evidence-2026-07-204","claude-mythos-5","Claude Mythos 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,88,88,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-632","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,83.7,83.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1965","claude-opus-4-7","Claude Opus 4.7","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,79.3,79.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-208","claude-opus-4-8","Claude Opus 4.8","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,84.3,84.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2057","claude-opus-5","Claude Opus 5","BrowseComp agentic search evaluation as published in the Anthropic Claude Opus 5 launch table.",null,"BrowseComp agentic search evaluation as published in the Anthropic Claude Opus 5 launch table.","browsecomp","BrowseComp","research","OpenAI",null,90.8,90.8,"percent","higher","1.6.0","ranking-eligible","direct","2026-07-24","2026-07-24","production::anthropic-claude-opus-5-launch","anthropic-claude-opus-5-launch","Introducing Claude Opus 5","Anthropic","https://www.anthropic.com/news/claude-opus-5","2026-07-24","2026-07-24","2026-07-24","provider-reported","Anthropic reports 90.8% on BrowseComp."],["evidence-2026-07-206","claude-sonnet-5","Claude Sonnet 5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,84.7,84.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1963","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,83.4,83.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1971","glm-4-7","GLM-4.7","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,52,52,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-633","glm-5-1","GLM-5.1","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,68,68,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1966","gpt-5-2","GPT-5.2","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,65.8,65.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-211","gpt-5-4","GPT-5.4","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,82.7,82.7,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-207","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,84.4,84.4,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-631","gpt-5-5","GPT-5.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","gpt-5-5-pro-default-high","Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,90.1,90.1,"percent","higher","1.3.0","reference-only","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-210","gpt-5-6-luna","GPT-5.6 Luna","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,83.3,83.3,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-203","gpt-5-6-sol","GPT-5.6 Sol","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,92.2,92.2,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-205","gpt-5-6-terra","GPT-5.6 Terra","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,87.5,87.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-212","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,60.6,60.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1964","kimi-k2-6","Kimi K2.6","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,83.2,83.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1937","kimi-k3","Kimi K3","BrowseComp with context compaction triggered at 300K tokens; max reasoning; as stated in the Kimi K3 launch blog.",null,"BrowseComp with context compaction triggered at 300K tokens; max reasoning; as stated in the Kimi K3 launch blog.","browsecomp","BrowseComp","research","OpenAI",null,91.2,91.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 91.2% with 300K compaction; 90.4% at 1M context without context management also cited."],["evidence-2026-07-1937--configuration--kimi-k3-max","kimi-k3","Kimi K3","BrowseComp with context compaction triggered at 300K tokens; max reasoning; as stated in the Kimi K3 launch blog.","kimi-k3-max","BrowseComp with context compaction triggered at 300K tokens; max reasoning; as stated in the Kimi K3 launch blog.","browsecomp","BrowseComp","research","OpenAI",null,91.2,91.2,"percent","higher","1.4.1","reference-only","direct","2026-07-16","2026-07-16","production::moonshot-kimi-k3-blog","moonshot-kimi-k3-blog","Kimi K3: Open Frontier Intelligence","Moonshot AI","https://www.kimi.com/blog/kimi-k3","2026-07-16","2026-07-21","2026-07-21","provider-reported","Moonshot reports 91.2% with 300K compaction; 90.4% at 1M context without context management also cited."],["evidence-2026-07-209","minimax-m3","MiniMax M3","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","browsecomp","BrowseComp","research","OpenAI",null,83.52,83.52,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1972","nemotron-3-ultra","Nemotron 3 Ultra","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,44.4,44.4,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["nvidia-nemotron-3-5-lightning-browsecomp-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","browsecomp","BrowseComp","research","OpenAI",null,36.97,36.97,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1967","qwen3-5-122b","Qwen3.5 122B","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,63.8,63.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1969","qwen3-5-27b","Qwen3.5 27B","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,61,61,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1970","qwen3-5-35b","Qwen3.5 35B","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,61,61,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1968","qwen3-5-397b","Qwen3.5 397B A17B","Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","browsecomp","BrowseComp","research","OpenAI",null,62,62,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-browsecomp-2026-07-20","benchlm-browsecomp-2026-07-20","BrowseComp Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/browsecomp","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public browsecomp leaderboard for the exact model variant. Independent-lab provenance."],["benchlm-ref-claude-opus-4-8-financeagentv2-2026-07-21","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",53.9,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-financeagentv2-2026-07-27","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",53.9,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-8-financeagentv2-2026-08-01","claude-opus-4-8","Claude Opus 4.8","Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",53.9,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-financeagentv2-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.861,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-financeagentv2-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.861,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-financeagentv2-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.861,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-financeagentv2-2026-07-21","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.2,83.3123,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-financeagentv2-2026-07-27","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.2,83.3123,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-1-1-financeagentv2-2026-08-01","muse-spark-1-1","Muse Spark 1.1","Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","finance-agent-v2","Finance Agent v2","research","Google","2026",57.2,83.3123,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-014","gemini-3-1-pro-preview","Gemini 3.1 Pro Preview","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","finance-agent-v2","Finance Agent v2","research","Google","v2",43,43,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-005","gemini-3-5-flash","Gemini 3.5 Flash","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","finance-agent-v2","Finance Agent v2","research","Google","v2",57.9,57.9,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-024","gpt-5-5","GPT-5.5","Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.",null,"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","finance-agent-v2","Finance Agent v2","research","Google","v2",51.8,51.8,"percent","higher","1.0.0","reference-only","direct","2026-05-19","2026-05-19","production::google-gemini-35-model-card","google-gemini-35-model-card","Gemini 3.5 Flash model card","Google DeepMind","https://deepmind.google/models/model-cards/gemini-3-5-flash/","2026-05-19","2026-07-15","2026-07-15","provider-reported","Exact provider-published evaluation table; retained with provider-reported provenance."],["evidence-2026-07-1907","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Finance Agent v2).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Finance Agent v2).","finance-agent-v2","Finance Agent v2","research","Google","v2",57.2,57.2,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Finance Agent v2; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1907--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Finance Agent v2).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Finance Agent v2).","finance-agent-v2","Finance Agent v2","research","Google","v2",57.2,57.2,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for Finance Agent v2; retained with provider-reported provenance via Meta evaluation report."],["benchlm-ref-claude-opus-4-6-healthbenchhard-2026-07-21","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",14.8,0,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-healthbenchhard-2026-07-27","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",14.8,0,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-claude-opus-4-6-healthbenchhard-2026-08-01","claude-opus-4-6","Claude Opus 4.6","Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",14.8,0,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-07-21","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.6,20.7143,"percent","higher","1.5.0","excluded","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-07-27","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.6,20.7143,"percent","higher","1.6.0","excluded","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-08-01","gemini-3-1-pro","Gemini 3.1 Pro","Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.6,20.7143,"percent","higher","1.8.0","excluded","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","excluded","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-healthbenchhard-2026-07-21","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",40.1,90.3571,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-healthbenchhard-2026-07-27","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",40.1,90.3571,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-4-healthbenchhard-2026-08-01","gpt-5-4","GPT-5.4","Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",40.1,90.3571,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-07-21","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32,61.4286,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-07-27","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32,61.4286,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-08-01","gpt-5-6-luna","GPT-5.6 Luna","Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32,61.4286,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-07-21","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",33.1,65.3571,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-07-27","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",33.1,65.3571,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-08-01","gpt-5-6-sol","GPT-5.6 Sol","Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",33.1,65.3571,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-07-21","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32.7,63.9286,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-07-27","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32.7,63.9286,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-08-01","gpt-5-6-terra","GPT-5.6 Terra","Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",32.7,63.9286,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-healthbenchhard-2026-07-21","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.3,19.6429,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-healthbenchhard-2026-07-27","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.3,19.6429,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-grok-4-20-beta-healthbenchhard-2026-08-01","grok-4-20","Grok 4.20","Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",20.3,19.6429,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-healthbenchhard-2026-07-21","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",42.8,100,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-healthbenchhard-2026-07-27","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",42.8,100,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-muse-spark-healthbenchhard-2026-08-01","muse-spark","Muse Spark","Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","healthbench-hard","HealthBench Hard","research","OpenAI","2026",42.8,100,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["google-gemini-37-eval-labbench2-claude-sonnet-5-2026-08-13","claude-sonnet-5","Claude Sonnet 5","Claude Sonnet 5","claude-sonnet-5-max","Claude Sonnet 5","labbench2","LABBench2","research","LABBench","2",80.1,80.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-labbench2-gemini-3-6-flash-2026-08-13","gemini-3-6-flash","Gemini 3.6 Flash","Gemini 3.6 Flash","gemini-3-6-flash-high","Gemini 3.6 Flash","labbench2","LABBench2","research","LABBench","2",76.1,76.1,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["google-gemini-37-eval-labbench2-gemini-3-7-flash-2026-08-13","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash","gemini-3-7-flash-medium","Gemini 3.7 Flash","labbench2","LABBench2","research","LABBench","2",82.1,82.1,"percent","higher","1.8.0","reference-only","direct","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Direct Google evaluation or owner-leaderboard value. Default sampling settings unless the methodology states otherwise."],["google-gemini-37-eval-labbench2-gpt-5-6-terra-2026-08-13","gpt-5-6-terra","GPT-5.6 Terra","GPT-5.6 Terra","gpt-5-6-terra-max","GPT-5.6 Terra","labbench2","LABBench2","research","LABBench","2",81.2,81.2,"percent","higher","1.8.0","reference-only","supported","2026-08-13","2026-08-13","production::refresh-google-gemini-3-7-flash-evaluation","refresh-google-gemini-3-7-flash-evaluation","Google DeepMind permanent refresh source","Google DeepMind","https://storage.googleapis.com/deepmind-media/gemini/gemini_3-7_flash_model_evaluation.pdf","2026-08-13","2026-08-15","2026-08-13","provider-reported","Google states that non-Gemini values are provider self-reports or best available public results; maximum reasoning is preferred where available."],["evidence-2026-08-15-command-a-plus-mmlu-pro-standard-table","command-a-plus","Command A+","Command A+ as published by Upstage","command-a-plus-solar-open2-unspecified","Command A+ as published by Upstage","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",79,79,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-deepseek-v4-flash-mmlu-pro-standard-table","deepseek-v4-flash","DeepSeek V4 Flash","DeepSeek V4 Flash max as published by Upstage","deepseek-v4-flash-max","DeepSeek V4 Flash max as published by Upstage","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",85.9,85.9,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mimo-v2-5-mmlu-pro-standard-table","mimo-v2-5","MiMo-V2.5","MiMo-V2.5 as published by Upstage","mimo-v2-5-solar-open2-unspecified","MiMo-V2.5 as published by Upstage","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",84.6,84.6,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-mistral-medium-3-5-mmlu-pro-standard-table","mistral-medium-3-5","Mistral Medium 3.5","Mistral Medium 3.5 as published by Upstage","mistral-medium-3-5-high","Mistral Medium 3.5 as published by Upstage","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",81.2,81.2,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open-100b-reasoning-mmlu-pro-standard-table","solar-open-100b-reasoning","Solar Open 100B (Reasoning)","Solar Open 100B as published by Upstage","solar-open-100b-reasoning-high","Solar Open 100B as published by Upstage","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",80.4,80.4,"percent","higher","2.1.0","excluded","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Competitor cell from the official comparison table; excluded from canonical scoring."],["evidence-2026-08-15-solar-open2-250b-mmlu-pro-standard","solar-open2-250b","Solar Open 2 250B","Solar Open 2 250B (high)","solar-open2-250b-high","Solar Open 2 250B (high)","mmlu-pro","MMLU-Pro","research","TIGER-Lab","standard",86.2,86.2,"percent","higher","2.1.0","ranking-eligible","direct","2026-08-15","2026-08-15","production::upstage-solar-open2-250b-huggingface-2026-08-15","upstage-solar-open2-250b-huggingface-2026-08-15","Solar Open 2 250B model card","Upstage","https://huggingface.co/upstage/Solar-Open2-250B","2026-08-15","2026-08-15","2026-08-15","provider-reported","Official owner cell from the complete published comparison table."],["evidence-2026-07-565","claude-opus-4-5","Claude Opus 4.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,89.5,89.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-569","claude-opus-4-6","Claude Opus 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,82,82,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-570","claude-sonnet-4-6","Claude Sonnet 4.6","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,79.2,79.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-274","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,83,83,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-275","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,82.9,82.9,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-571","gemini-3-pro","Gemini 3 Pro Preview","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,89.8,89.8,"percent","higher","1.3.0","excluded","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","excluded","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1982","gemma-4-12b","Gemma 4 12B Unified","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,77.2,77.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1981","gemma-4-26b","Gemma 4 26B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,82.6,82.6,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-568","gemma-4-31b","Gemma 4 31B","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,85.2,85.2,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-1980","glm-4-7","GLM-4.7","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,84.3,84.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-567","glm-5","GLM-5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,85.7,85.7,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-273","kimi-k2-5","Kimi K2.5","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,87.1,87.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1979","mimo-v2-flash","MiMo-V2-Flash","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,84.9,84.9,"percent","higher","1.4.1","excluded","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","excluded","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1974","nemotron-3-ultra","Nemotron 3 Ultra","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,86.8,86.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["nvidia-nemotron-3-5-lightning-mmlu-pro-2026-08-11","nemotron-3-5-lightning-30b-a3b","NVIDIA Nemotron 3.5 Lightning 30B-A3B","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","nemotron-3-5-lightning-30b-a3b-default","NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16; NVIDIA release evaluation; temperature=1.0; top_p=0.95","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,81.94,81.94,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-11","2026-08-11","production::nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","nvidia-nemotron-3-5-lightning-model-card-ce38b6ab","NVIDIA Nemotron 3.5 Lightning 30B-A3B BF16 model card","NVIDIA","https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16/blob/ce38b6ab8b252b4b8ee7165b4605e93191cafd73/README.md","2026-08-11","2026-08-12","2026-08-12","provider-reported","Exact NVIDIA model-card value. NVIDIA states that its consistent-harness accuracy numbers may differ from vendors' self-reported results."],["evidence-2026-07-1975","qwen3-5-122b","Qwen3.5 122B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,86.7,86.7,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1977","qwen3-5-27b","Qwen3.5 27B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,86.1,86.1,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1978","qwen3-5-35b","Qwen3.5 35B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,85.3,85.3,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1973","qwen3-5-397b","Qwen3.5 397B A17B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,87.8,87.8,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-1976","qwen3-6-27b","Qwen3.6 27B","Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).",null,"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,86.2,86.2,"percent","higher","1.4.1","ranking-eligible","direct","2026-07-20","2026-07-20","production::benchlm-mmlu-pro-2026-07-20","benchlm-mmlu-pro-2026-07-20","MMLU-Pro Leaderboard & Scores — July 2026","BenchLM","https://benchlm.ai/benchmarks/mmluPro","2026-07-20","2026-07-21","2026-07-21","source-checked","Mirrored from BenchLM public mmlu-pro leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-566","qwen3-6-plus","Qwen3.6 Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,88.5,88.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public leaderboard only. No synthetic or invented values."],["evidence-2026-07-271","qwen-3-7-max","Qwen3.7-Max","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,89.6,89.6,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-272","qwen-3-7-plus","Qwen3.7-Plus","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","mmlu-pro","MMLU-Pro","research","TIGER-Lab",null,88.5,88.5,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-2052","gemini-3-1-flash-lite","Gemini 3.1 Flash-Lite","As published in Gemini 3.5 Flash-Lite launch materials.",null,"As published in Gemini 3.5 Flash-Lite launch materials.","mrcr-v2","MRCRv2","research","OpenAI","2",60.1,60.1,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","GDM-MRCR v2 60.1% for 3.1 Flash-Lite."],["evidence-2026-07-2035","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDM-MRCR v2 as published in Google launch blog.",null,"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDM-MRCR v2 as published in Google launch blog.","mrcr-v2","MRCRv2","research","OpenAI","2",72.2,72.2,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-blog","google-gemini-36-blog","Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","Google","https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","2026-07-21","2026-07-21","2026-07-21","provider-reported","GDM-MRCR v2 72.2% for 3.5 Flash-Lite."],["evidence-2026-07-1996","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 cumulative score at 128k context.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 cumulative score at 128k context.","mrcr-v2","MRCRv2","research","OpenAI","2",91.8,91.8,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","GDM-MRCR v2 128k average from Google evaluation table."],["evidence-2026-07-1997","gemini-3-6-flash","Gemini 3.6 Flash","gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 pointwise score at 1M context.",null,"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 pointwise score at 1M context.","mrcr-v2","MRCRv2","research","OpenAI","2",54,54,"percent","higher","1.4.1","reference-only","direct","2026-07-21","2026-07-21","production::google-gemini-36-product","google-gemini-36-product","Gemini 3.6 Flash product page with evaluation table","Google DeepMind","https://deepmind.google/models/gemini/flash/","2026-07-21","2026-07-21","2026-07-21","provider-reported","GDM-MRCR v2 1M pointwise from Google evaluation table."],["evidence-2026-07-1913","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MRCR 1M).",null,"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MRCR 1M).","mrcr-v2","MRCRv2","research","OpenAI","2",54.1,54.1,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for MRCR 1M; retained with provider-reported provenance via Meta evaluation report."],["evidence-2026-07-1913--configuration--muse-spark-1-1-xhigh","muse-spark-1-1","Muse Spark 1.1","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MRCR 1M).","muse-spark-1-1-xhigh","Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MRCR 1M).","mrcr-v2","MRCRv2","research","OpenAI","2",54.1,54.1,"percent","higher","1.4.1","reference-only","direct","2026-07-09","2026-07-09","production::meta-muse-spark-1-1-eval","meta-muse-spark-1-1-eval","Meta AI Muse Spark 1.1 evaluation report","Meta","https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","2026-07-09","2026-07-16","2026-07-16","provider-reported","Exact provider-published value for MRCR 1M; retained with provider-reported provenance via Meta evaluation report."],["benchlm-ref-gemini-3-5-flash-mrcrv2-2026-07-21","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",77.3,67.5299,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mrcrv2-2026-07-27","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",77.3,67.5299,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-mrcrv2-2026-08-01","gemini-3-5-flash","Gemini 3.5 Flash","Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",77.3,67.5299,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-07-21","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",72.2,57.3705,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-07-27","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",72.2,57.3705,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-08-01","gemini-3-5-flash-lite","Gemini 3.5 Flash-Lite","Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",72.2,57.3705,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mrcrv2-2026-07-21","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",43.4,0,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mrcrv2-2026-07-27","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",43.4,0,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-gemma-4-12b-mrcrv2-2026-08-01","gemma-4-12b","Gemma 4 12B Unified","Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",43.4,0,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mrcrv2-2026-07-21","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",90.4,93.6255,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mrcrv2-2026-07-27","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",90.4,93.6255,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-max-mrcrv2-2026-08-01","qwen-3-7-max","Qwen3.7-Max","Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",90.4,93.6255,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mrcrv2-2026-07-21","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",91.7,96.2151,"percent","higher","1.5.0","reference-only","supported","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mrcrv2-2026-07-27","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",91.7,96.2151,"percent","higher","1.6.0","reference-only","supported","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-qwen3-7-plus-mrcrv2-2026-08-01","qwen-3-7-plus","Qwen3.7-Plus","Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",91.7,96.2151,"percent","higher","1.8.0","reference-only","supported","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-mrcrv2-2026-07-21","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",86.6,86.0558,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-mrcrv2-2026-07-27","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",86.6,86.0558,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-mrcrv2-2026-08-01","sakana-fugu","Sakana Fugu","Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",86.6,86.0558,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-07-21","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",93.6,100,"percent","higher","1.5.0","reference-only","estimated","2026-07-21","2026-07-21","production::benchlm-public-dataset-2026-07-21","benchlm-public-dataset-2026-07-21","BenchLM public datasets — 21 July 2026","BenchLM","https://benchlm.ai/embed","2026-07-21","2026-07-21","2026-07-21","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-07-27","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",93.6,100,"percent","higher","1.6.0","reference-only","estimated","2026-07-27","2026-07-27","production::benchlm-public-dataset-2026-07-27","benchlm-public-dataset-2026-07-27","BenchLM public datasets — 2026-07-27","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-07-27","2026-07-27","2026-07-27","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-08-01","sakana-fugu-ultra","Sakana Fugu-Ultra","Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.",null,"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","mrcr-v2","MRCRv2","research","OpenAI","2025",93.6,100,"percent","higher","1.8.0","reference-only","estimated","2026-08-01","2026-08-01","production::benchlm-public-dataset-2026-08-01","benchlm-public-dataset-2026-08-01","BenchLM public datasets — 2026-08-01","BenchLM","https://benchlm.ai/data/leaderboard.json","2026-08-01","2026-08-01","2026-08-01","source-checked","Reference-only public registry row. Not eligible for Lumina scoring without exact upstream configuration."],["evidence-2026-07-869","kimi-k2-5","Kimi K2.5","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","paperbench","PaperBench","research","OpenAI",null,63.5,63.5,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::llm-stats-paperbench","llm-stats-paperbench","PaperBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/paperbench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-07-870","minimax-m3","MiniMax M3","Exact model variant as published on the source leaderboard page.",null,"Exact model variant as published on the source leaderboard page.","paperbench","PaperBench","research","OpenAI",null,52.6,52.6,"percent","higher","1.3.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::llm-stats-paperbench","llm-stats-paperbench","PaperBench scores via LLM Stats","LLM Stats","https://llm-stats.com/benchmarks/paperbench","2026-07-15","2026-07-15","2026-07-15","source-checked","Imported from public leaderboard for the exact named model variant only. No invented or remapped non-matching variants."],["evidence-2026-08-qwen-3-8-max-paperbench","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)",null,"Qwen3.8 Max (xhigh, default reasoning effort)","paperbench","PaperBench","research","OpenAI",null,93,93,"percent","higher","1.8.0","ranking-eligible","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 93.0. Qwen used BasicAgent in Code-Dev mode, a Claude Opus 4.6 judge, three-run averaging, and a 12-hour maximum."],["evidence-2026-08-qwen-3-8-max-paperbench--configuration--qwen-3-8-max-xhigh","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh, default reasoning effort)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh, default reasoning effort)","paperbench","PaperBench","research","OpenAI",null,93,93,"percent","higher","1.8.0","reference-only","direct","2026-08-03","2026-08-03","production::qwen-qwen38-release-2026-08-03","qwen-qwen38-release-2026-08-03","Qwen3.8 Max official release","Qwen","https://qwen.ai/blog?id=qwen3.8","2026-08-03","2026-08-05","2026-08-05","provider-reported","Official Qwen result: 93.0. Qwen used BasicAgent in Code-Dev mode, a Claude Opus 4.6 judge, three-run averaging, and a 12-hour maximum."],["evidence-2026-08-15-longcat-2-0-rwsearch-2026-08","longcat-2-0","LongCat-2.0","LongCat-2.0 (in-house harness)","longcat-2-0-default","LongCat-2.0 (in-house harness)","rwsearch","RWSearch","research","AGI-Eval-Official","2026-08",78.8,78.8,"percent","higher","2.1.0","reference-only","direct","2026-06-30","2026-06-30","production::longcat-2-0-huggingface-2026-08-15","longcat-2-0-huggingface-2026-08-15","LongCat-2.0 model card","Meituan LongCat","https://huggingface.co/meituan-longcat/LongCat-2.0","2026-06-30","2026-08-29","2026-08-15","provider-reported","Official LongCat-2.0 in-house RWSearch score."],["evidence-2026-07-270","deepseek-v4-flash","DeepSeek V4 Flash","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","simpleqa","SimpleQA","research","OpenAI",null,23.1,23.1,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["evidence-2026-07-269","deepseek-v4-pro","DeepSeek V4 Pro","Exact model variant as listed on the BenchLM public benchmark leaderboard page.",null,"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","simpleqa","SimpleQA","research","OpenAI",null,45,45,"percent","higher","1.2.0","ranking-eligible","direct","2026-07-15","2026-07-15","production::benchlm-public-benchmarks","benchlm-public-benchmarks","BenchLM public benchmark leaderboards","BenchLM","https://benchlm.ai/benchmarks","2026-07-15","2026-07-16","2026-07-15","source-checked","Mirrored from BenchLM public benchmark leaderboard for the exact model variant. Independent-lab provenance."],["epoch-simpleqa-verified-84ARuGyhQsjgh8umRmBvas","gemini-3-7-flash","Gemini 3.7 Flash","Gemini 3.7 Flash (high)","gemini-3-7-flash-high","Gemini 3.7 Flash (high)","simpleqa-verified","SimpleQA Verified","research","OpenAI / Google","1.2.0",69.2,69.2,"percent","higher","2.2.0","reference-only","direct","2026-08-27","2026-08-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-27","2026-09-01","2026-09-01","independently-verified","simpleqa_scorer:0.69±0.01; stderr=0.01460648312734278; Epoch run 84ARuGyhQsjgh8umRmBvas. SimpleQA Verified 1.2.0 is retained as exact public reference evidence because its anti-abstention protocol is not a frozen scoring family."],["epoch-simpleqa-verified-TY8R8FV7zCwByEipkqyg2P","glm-5-3","GLM-5.3","GLM-5.3 (max)","glm-5-3-max","GLM-5.3 (max)","simpleqa-verified","SimpleQA Verified","research","OpenAI / Google","1.2.0",41,41,"percent","higher","2.2.0","reference-only","direct","2026-08-28","2026-08-28","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-28","2026-09-01","2026-09-01","independently-verified","simpleqa_scorer:0.41±0.02; stderr=0.015560917136921659; Epoch run TY8R8FV7zCwByEipkqyg2P. SimpleQA Verified 1.2.0 is retained as exact public reference evidence because its anti-abstention protocol is not a frozen scoring family."],["epoch-simpleqa-verified-dKCWFaoGCWJTnD5CPWQGru","grok-4-6","Grok 4.6","Grok 4.6 (xhigh)","grok-4-6-xhigh","Grok 4.6 (xhigh)","simpleqa-verified","SimpleQA Verified","research","OpenAI / Google","1.2.0",48.9,48.9,"percent","higher","2.2.0","reference-only","direct","2026-08-27","2026-08-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-27","2026-09-01","2026-09-01","independently-verified","simpleqa_scorer:0.49±0.02; stderr=0.015815471195292575; Epoch run dKCWFaoGCWJTnD5CPWQGru. SimpleQA Verified 1.2.0 is retained as exact public reference evidence because its anti-abstention protocol is not a frozen scoring family."],["epoch-simpleqa-verified-mLLmVVvo6t23MM5yxq7ryW","muse-spark-1-2","Muse Spark 1.2","Muse Spark 1.2 (xhigh)","muse-spark-1-2-xhigh","Muse Spark 1.2 (xhigh)","simpleqa-verified","SimpleQA Verified","research","OpenAI / Google","1.2.0",60.301508,60.301508,"percent","higher","2.2.0","reference-only","direct","2026-08-27","2026-08-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-27","2026-09-01","2026-09-01","independently-verified","simpleqa_scorer:0.6±0.02; stderr=0.015518791563606454; Epoch run mLLmVVvo6t23MM5yxq7ryW. SimpleQA Verified 1.2.0 is retained as exact public reference evidence because its anti-abstention protocol is not a frozen scoring family."],["epoch-simpleqa-verified-bfftd5EzUK9yAekafEofvK","qwen-3-8-max","Qwen3.8 Max","Qwen3.8 Max (xhigh)","qwen-3-8-max-xhigh","Qwen3.8 Max (xhigh)","simpleqa-verified","SimpleQA Verified","research","OpenAI / Google","1.2.0",45.8,45.8,"percent","higher","2.2.0","reference-only","direct","2026-08-27","2026-08-27","production::refresh-epoch-benchmarks","refresh-epoch-benchmarks","Epoch AI permanent refresh source","Epoch AI","https://epoch.ai/data/benchmarks.csv","2026-08-27","2026-09-01","2026-09-01","independently-verified","simpleqa_scorer:0.46±0.02; stderr=0.015763390640483554; Epoch run bfftd5EzUK9yAekafEofvK. SimpleQA Verified 1.2.0 is retained as exact public reference evidence because its anti-abstention protocol is not a frozen scoring family."]],"stableKey":"resultId","table":"benchmark-results"}
